table-validator 0.1.4__tar.gz → 0.1.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {table_validator-0.1.4/table_validator.egg-info → table_validator-0.1.6}/PKG-INFO +1 -1
  2. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/catalog_validator.py +36 -9
  3. {table_validator-0.1.4 → table_validator-0.1.6/table_validator.egg-info}/PKG-INFO +1 -1
  4. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/scm_version.json +2 -2
  5. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_catalog_validator.py +72 -0
  6. {table_validator-0.1.4 → table_validator-0.1.6}/LICENSE +0 -0
  7. {table_validator-0.1.4 → table_validator-0.1.6}/README.md +0 -0
  8. {table_validator-0.1.4 → table_validator-0.1.6}/pyproject.toml +0 -0
  9. {table_validator-0.1.4 → table_validator-0.1.6}/setup.cfg +0 -0
  10. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/__init__.py +0 -0
  11. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/auth/__init__.py +0 -0
  12. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/auth/azure_auth.py +0 -0
  13. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/auth/databricks_auth.py +0 -0
  14. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/__init__.py +0 -0
  15. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/main.py +0 -0
  16. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/partition_prompt.py +0 -0
  17. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/summary_table.py +0 -0
  18. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/cli/wizard.py +0 -0
  19. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/config/__init__.py +0 -0
  20. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/config/manager.py +0 -0
  21. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/config/schema.py +0 -0
  22. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/__init__.py +0 -0
  23. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/azure_connector.py +0 -0
  24. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/connectors/databricks_connector.py +0 -0
  25. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/engine/__init__.py +0 -0
  26. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/engine/comparison_engine.py +0 -0
  27. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/models.py +0 -0
  28. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/reports/__init__.py +0 -0
  29. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/reports/excel_report.py +0 -0
  30. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/__init__.py +0 -0
  31. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/blob_discovery.py +0 -0
  32. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator/validators/row_validator.py +0 -0
  33. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/SOURCES.txt +0 -0
  34. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/dependency_links.txt +0 -0
  35. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/entry_points.txt +0 -0
  36. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/requires.txt +0 -0
  37. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/scm_file_list.json +0 -0
  38. {table_validator-0.1.4 → table_validator-0.1.6}/table_validator.egg-info/top_level.txt +0 -0
  39. {table_validator-0.1.4 → table_validator-0.1.6}/tests/__init__.py +0 -0
  40. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_blob_discovery.py +0 -0
  41. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_cli.py +0 -0
  42. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_databricks_connector.py +0 -0
  43. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_excel_report.py +0 -0
  44. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_partition_prompt.py +0 -0
  45. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_report_command.py +0 -0
  46. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_row_validator.py +0 -0
  47. {table_validator-0.1.4 → table_validator-0.1.6}/tests/test_wizard.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.4
3
+ Version: 0.1.6
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -398,6 +398,18 @@ class CatalogValidator:
398
398
  if pair_key in existing_pairs_lower:
399
399
  continue
400
400
 
401
+ # An explicit mapping for this source schema supersedes any
402
+ # identical-name pair already seeded for it (e.g. a
403
+ # coincidental same-named schema on both sides) - the user
404
+ # named exactly one target for this source schema, so it must
405
+ # be validated once, against that target, not once against
406
+ # its own identical-name match AND again against the mapped
407
+ # target.
408
+ common_pairs = [
409
+ (s, t) for s, t in common_pairs if s.lower() != actual_source.lower()
410
+ ]
411
+ existing_pairs_lower = {(s.lower(), t.lower()) for s, t in common_pairs}
412
+
401
413
  common_pairs.append((actual_source, actual_target))
402
414
  existing_pairs_lower.add(pair_key)
403
415
  missing_set.discard(actual_source)
@@ -516,15 +528,18 @@ class CatalogValidator:
516
528
  names within an already-resolved source_schema/target_schema
517
529
  pair).
518
530
  """
519
- common, missing, extra = self.compare_tables(
520
- request.source_catalog, request.target_catalog, source_schema
521
- )
522
-
523
- # NOTE: compare_tables() above assumes the same schema name on
524
- # both sides. When source_schema != target_schema (a schema_map
525
- # pair), that identical-name baseline is meaningless - re-derive
526
- # it directly from each side's own schema.
527
- if source_schema.lower() != target_schema.lower():
531
+ if source_schema.lower() == target_schema.lower():
532
+ common, missing, extra = self.compare_tables(
533
+ request.source_catalog, request.target_catalog, source_schema
534
+ )
535
+ else:
536
+ # compare_tables() assumes the same schema name on both
537
+ # sides - calling it here would query the TARGET catalog for
538
+ # a schema named after the SOURCE schema (a schema_map pair
539
+ # means these differ), which doesn't exist there and raises
540
+ # SCHEMA_NOT_FOUND instead of just being a wasted query.
541
+ # Derive the baseline directly from each side's own,
542
+ # correctly-named schema instead.
528
543
  try:
529
544
  source_tables = set(self.databricks.get_tables(request.source_catalog, source_schema))
530
545
  except Exception:
@@ -586,6 +601,18 @@ class CatalogValidator:
586
601
  if pair_key in existing_pairs_lower:
587
602
  continue
588
603
 
604
+ # An explicit mapping for this source table supersedes any
605
+ # identical-name pair already seeded for it (e.g. a
606
+ # coincidental same-named table on both sides) - the user
607
+ # named exactly one target for this source table, so it must
608
+ # be validated once, against that target, not once against
609
+ # its own identical-name match AND again against the mapped
610
+ # target.
611
+ common_pairs = [
612
+ (s, t) for s, t in common_pairs if s.lower() != actual_source.lower()
613
+ ]
614
+ existing_pairs_lower = {(s.lower(), t.lower()) for s, t in common_pairs}
615
+
589
616
  common_pairs.append((actual_source, actual_target))
590
617
  existing_pairs_lower.add(pair_key)
591
618
  missing_set.discard(actual_source)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.4
3
+ Version: 0.1.6
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -1,7 +1,7 @@
1
1
  {
2
- "tag": "0.1.4",
2
+ "tag": "0.1.6",
3
3
  "distance": 0,
4
- "node": "gb4d33e041e4eaea0e6645d64709a7fe7142c76db",
4
+ "node": "g9357213ed68fd98fc2c232aed2bb737f273cabbf",
5
5
  "dirty": false,
6
6
  "branch": "HEAD",
7
7
  "node_date": "2026-08-28"
@@ -1473,6 +1473,36 @@ def test_table_map_validates_explicit_pair_with_different_names():
1473
1473
  connector.get_row_count.assert_called()
1474
1474
 
1475
1475
 
1476
+ def test_table_map_source_name_existing_identically_on_target_is_not_validated_twice():
1477
+ """Real bug: the source table ('jd_example_data_2') coincidentally also
1478
+ exists under that exact same name on the target side (an unrelated
1479
+ leftover table), so compare_tables' plain intersection legitimately
1480
+ finds an identical-name pair for it - in ADDITION to the table_map
1481
+ entry pairing it with a different target ('jd_example_data_90'). The
1482
+ mapping must supersede the identical-name match, not run alongside
1483
+ it: one source table configured by the user must produce exactly one
1484
+ validated table, against the mapped target only."""
1485
+ connector = _make_connector()
1486
+ connector.get_tables.side_effect = lambda catalog, schema: (
1487
+ ["jd_example_data_2"] if catalog == "cat_source"
1488
+ else ["jd_example_data_2", "jd_example_data_90"]
1489
+ )
1490
+ validator = CatalogValidator(connector)
1491
+
1492
+ result = validator.compare_catalogs(
1493
+ _request(
1494
+ tables=["jd_example_data_2"],
1495
+ table_map={"jd_example_data_2": "jd_example_data_90"},
1496
+ )
1497
+ )
1498
+
1499
+ schema_result = result.schemas[0]
1500
+ assert len(schema_result.tables) == 1
1501
+ table = schema_result.tables[0]
1502
+ assert table.table == "jd_example_data_90"
1503
+ assert table.source_table_name == "jd_example_data_2"
1504
+
1505
+
1476
1506
  def test_table_map_to_nonexistent_target_table_produces_clear_error():
1477
1507
  """A table_map entry naming a target table that does NOT exist must
1478
1508
  produce a visible FAIL/ERROR with an informative message - never a
@@ -1544,6 +1574,48 @@ def test_schema_map_to_nonexistent_target_schema_produces_clear_error():
1544
1574
  assert result.status in (ValidationStatus.ERROR, ValidationStatus.FAIL)
1545
1575
 
1546
1576
 
1577
+ def test_schema_map_does_not_query_target_using_source_schema_name():
1578
+ """Real bug: _resolve_table_pairs unconditionally called compare_tables
1579
+ with the SOURCE schema name for both catalogs before checking whether
1580
+ schema_map applied - querying the target catalog for a schema named
1581
+ after the source schema, which doesn't exist there under a real
1582
+ Databricks connector (raises SCHEMA_NOT_FOUND) rather than just being
1583
+ a wasted no-op query. get_tables must only ever be called with each
1584
+ catalog's own correctly-mapped schema name."""
1585
+ connector = _make_connector()
1586
+ connector.get_schemas.side_effect = lambda catalog: (
1587
+ ["for_schema_validation"] if catalog == "cat_source" else ["bronze"]
1588
+ )
1589
+
1590
+ def get_tables(catalog, schema):
1591
+ if catalog == "cat_source" and schema == "for_schema_validation":
1592
+ return ["bronze_fn_sku_add"]
1593
+ if catalog == "cat_target" and schema == "bronze":
1594
+ return ["bronze_fn_sku_add"]
1595
+ raise RuntimeError(
1596
+ f"Unable to list tables for '{catalog}.{schema}': SCHEMA_NOT_FOUND"
1597
+ )
1598
+
1599
+ connector.get_tables.side_effect = get_tables
1600
+ validator = CatalogValidator(connector)
1601
+
1602
+ result = validator.compare_catalogs(
1603
+ _request(
1604
+ schemas=["for_schema_validation"],
1605
+ schema_map={"for_schema_validation": "bronze"},
1606
+ tables=["bronze_fn_sku_add"],
1607
+ )
1608
+ )
1609
+
1610
+ assert result.status != ValidationStatus.ERROR
1611
+ schema_result = result.schemas[0]
1612
+ assert schema_result.schema_name == "bronze"
1613
+ assert schema_result.status != ValidationStatus.ERROR
1614
+ assert len(schema_result.tables) == 1
1615
+ assert schema_result.tables[0].table == "bronze_fn_sku_add"
1616
+ assert schema_result.tables[0].status in (ValidationStatus.PASS, ValidationStatus.FAIL)
1617
+
1618
+
1547
1619
  def test_no_map_configured_identical_name_discovery_unchanged():
1548
1620
  """Regression guard: with no schema_map/table_map set (the default,
1549
1621
  and today's only behavior), discovery must be completely unchanged -
File without changes