table-validator 0.1.1__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {table_validator-0.1.1/table_validator.egg-info → table_validator-0.1.2}/PKG-INFO +1 -1
  2. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/validators/catalog_validator.py +29 -18
  3. {table_validator-0.1.1 → table_validator-0.1.2/table_validator.egg-info}/PKG-INFO +1 -1
  4. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator.egg-info/scm_version.json +2 -2
  5. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_catalog_validator.py +40 -0
  6. {table_validator-0.1.1 → table_validator-0.1.2}/LICENSE +0 -0
  7. {table_validator-0.1.1 → table_validator-0.1.2}/README.md +0 -0
  8. {table_validator-0.1.1 → table_validator-0.1.2}/pyproject.toml +0 -0
  9. {table_validator-0.1.1 → table_validator-0.1.2}/setup.cfg +0 -0
  10. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/__init__.py +0 -0
  11. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/auth/__init__.py +0 -0
  12. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/auth/azure_auth.py +0 -0
  13. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/auth/databricks_auth.py +0 -0
  14. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/cli/__init__.py +0 -0
  15. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/cli/main.py +0 -0
  16. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/cli/partition_prompt.py +0 -0
  17. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/cli/summary_table.py +0 -0
  18. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/cli/wizard.py +0 -0
  19. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/config/__init__.py +0 -0
  20. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/config/manager.py +0 -0
  21. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/config/schema.py +0 -0
  22. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/connectors/__init__.py +0 -0
  23. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/connectors/azure_connector.py +0 -0
  24. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/connectors/databricks_connector.py +0 -0
  25. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/engine/__init__.py +0 -0
  26. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/engine/comparison_engine.py +0 -0
  27. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/models.py +0 -0
  28. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/reports/__init__.py +0 -0
  29. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/reports/excel_report.py +0 -0
  30. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/validators/__init__.py +0 -0
  31. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/validators/blob_discovery.py +0 -0
  32. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator/validators/row_validator.py +0 -0
  33. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator.egg-info/SOURCES.txt +0 -0
  34. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator.egg-info/dependency_links.txt +0 -0
  35. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator.egg-info/entry_points.txt +0 -0
  36. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator.egg-info/requires.txt +0 -0
  37. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator.egg-info/scm_file_list.json +0 -0
  38. {table_validator-0.1.1 → table_validator-0.1.2}/table_validator.egg-info/top_level.txt +0 -0
  39. {table_validator-0.1.1 → table_validator-0.1.2}/tests/__init__.py +0 -0
  40. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_blob_discovery.py +0 -0
  41. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_cli.py +0 -0
  42. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_databricks_connector.py +0 -0
  43. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_excel_report.py +0 -0
  44. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_partition_prompt.py +0 -0
  45. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_report_command.py +0 -0
  46. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_row_validator.py +0 -0
  47. {table_validator-0.1.1 → table_validator-0.1.2}/tests/test_wizard.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -88,6 +88,30 @@ class CatalogValidator:
88
88
  self.partition_prompt = partition_prompt
89
89
  logger.debug("CatalogValidator initialised")
90
90
 
91
+ @staticmethod
92
+ def _lookup_primary_key(
93
+ request: "CatalogValidationRequest",
94
+ schema_name: str,
95
+ table_name: str,
96
+ ) -> Optional[List[str]]:
97
+ """
98
+ Look up request.primary_keys (keyed by "schema.table" or bare
99
+ table name, as typed by the user into config.yaml/the wizard) for
100
+ the given schema_name/table_name.
101
+
102
+ schema_name/table_name here come from whatever casing Databricks'
103
+ information_schema actually returns, matched case-insensitively
104
+ against the catalog in compare_schemas/compare_tables - which
105
+ does not necessarily match the exact casing the user configured.
106
+ An exact-match dict lookup would silently miss a configured key
107
+ and fall through to the much more expensive ROW_NUMBER() fallback
108
+ with no indication why, so this normalizes both sides to
109
+ lowercase before comparing.
110
+ """
111
+ lowered = {k.lower(): v for k, v in request.primary_keys.items()}
112
+ key_lookup = f"{schema_name}.{table_name}".lower()
113
+ return lowered.get(key_lookup) or lowered.get(table_name.lower())
114
+
91
115
  # ------------------------------------------------------------------
92
116
  # Stage 1 + top-level orchestration
93
117
  # ------------------------------------------------------------------
@@ -697,10 +721,7 @@ class CatalogValidator:
697
721
 
698
722
  # Configured PK column missing from either side is also BLOCKING -
699
723
  # every later tier depends on being able to resolve a usable key.
700
- key_lookup = f"{schema_name}.{table_name}"
701
- configured_key = request.primary_keys.get(key_lookup) or request.primary_keys.get(
702
- table_name
703
- )
724
+ configured_key = self._lookup_primary_key(request, schema_name, table_name)
704
725
  if configured_key:
705
726
  common_lower = {c.lower() for c in common_cols}
706
727
  if any(k.lower() not in common_lower for k in configured_key):
@@ -1040,8 +1061,7 @@ class CatalogValidator:
1040
1061
  )
1041
1062
  return
1042
1063
 
1043
- key_lookup = f"{schema_name}.{table_name}"
1044
- key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(table_name)
1064
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1045
1065
  candidates = self._partition_candidates(common_cols, key_columns)
1046
1066
 
1047
1067
  context = PartitionPromptContext(
@@ -1184,10 +1204,7 @@ class CatalogValidator:
1184
1204
  result.tier_reached = ValidationTier.ROW_HASH
1185
1205
 
1186
1206
  if not using_row_number_fallback:
1187
- key_lookup = f"{schema_name}.{table_name}"
1188
- key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
1189
- table_name
1190
- )
1207
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1191
1208
  mismatched_keys = [m.primary_key for m in mismatches if m.status == "MISMATCH"]
1192
1209
  if key_columns and mismatched_keys:
1193
1210
  self._tier5_column_diff(
@@ -1238,10 +1255,7 @@ class CatalogValidator:
1238
1255
  written straight to `result.data`, so multiple bucket calls
1239
1256
  aggregate instead of each overwriting the last.
1240
1257
  """
1241
- key_lookup = f"{schema_name}.{table_name}"
1242
- row_hash_key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
1243
- table_name
1244
- )
1258
+ row_hash_key_columns = self._lookup_primary_key(request, schema_name, table_name)
1245
1259
 
1246
1260
  using_row_number_fallback = not row_hash_key_columns
1247
1261
  if using_row_number_fallback:
@@ -1470,10 +1484,7 @@ class CatalogValidator:
1470
1484
  ) -> DataValidationResult:
1471
1485
 
1472
1486
  mode = request.data_compare_mode
1473
- key = f"{schema_name}.{table_name}"
1474
- key_columns = request.primary_keys.get(key) or request.primary_keys.get(
1475
- table_name
1476
- )
1487
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1477
1488
 
1478
1489
  logger.info(
1479
1490
  "[compare_data] table=%s.%s | mode=%s | resolved_key_columns=%s",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -1,7 +1,7 @@
1
1
  {
2
- "tag": "0.1.1",
2
+ "tag": "0.1.2",
3
3
  "distance": 0,
4
- "node": "gd0c901c9dfc750527e33620c5dea6b15e43e4443",
4
+ "node": "g43988f9f51a097e239c49384ab59f00fec5851ab",
5
5
  "dirty": false,
6
6
  "branch": "HEAD",
7
7
  "node_date": "2026-08-27"
@@ -991,6 +991,46 @@ def test_tier5_column_diff_runs_for_mismatched_key_and_names_exact_column():
991
991
  assert detail.verified is True
992
992
 
993
993
 
994
+ def test_primary_key_lookup_is_case_insensitive_against_catalog_metadata():
995
+ """request.primary_keys is keyed by whatever casing the user typed
996
+ into config.yaml/the wizard, but schema_name/table_name at lookup
997
+ time come from Databricks' information_schema (matched
998
+ case-insensitively against the catalog elsewhere in this file). A
999
+ primary key configured as "Bronze.Customers" must still be found for
1000
+ the real "bronze.customers" table - an exact-match lookup would
1001
+ silently miss it and fall through to the much slower ROW_NUMBER()
1002
+ fallback instead of using the real configured key."""
1003
+ connector = _mismatched_fingerprint_connector()
1004
+ connector.get_row_hashes.side_effect = lambda catalog, schema, table, cols, pk, bucket_predicate=None: (
1005
+ _hash_df([(1, "aaa"), (2, "bbb")])
1006
+ if catalog == "cat_source"
1007
+ else _hash_df([(1, "aaa"), (2, "zzz")])
1008
+ )
1009
+ connector.get_row_detail_for_keys.return_value = [
1010
+ {
1011
+ "key": {"id": 2},
1012
+ "mismatched_columns": ["name"],
1013
+ "source_values": {"name": "old-value"},
1014
+ "target_values": {"name": "new-value"},
1015
+ "source_row_hash": "bbb",
1016
+ "target_row_hash": "zzz",
1017
+ }
1018
+ ]
1019
+ validator = CatalogValidator(connector)
1020
+
1021
+ result = validator.compare_catalogs(
1022
+ _request(primary_keys={"Bronze.Customers": ["id"]})
1023
+ )
1024
+ table = result.schemas[0].tables[0]
1025
+
1026
+ connector.get_row_hashes_by_row_number.assert_not_called()
1027
+ assert connector.get_row_hashes.call_count == 2 # once per side (source, target)
1028
+ connector.get_row_detail_for_keys.assert_called_once()
1029
+ assert table.data.key_columns == ["id"]
1030
+ assert table.tier_reached == ValidationTier.COLUMN_DIFF
1031
+ assert table.data.sample_changed_detail[0].verified is True
1032
+
1033
+
994
1034
  def test_tier5_column_diff_runs_for_row_number_fallback_when_mismatched():
995
1035
  """No primary key configured -> row-number fallback is used for Tier
996
1036
  4, but a real mismatch must still let Tier 5 attempt best-effort
File without changes