table-validator 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {table_validator-0.1.0/table_validator.egg-info → table_validator-0.1.2}/PKG-INFO +1 -1
  2. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/main.py +29 -6
  3. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/catalog_validator.py +29 -18
  4. {table_validator-0.1.0 → table_validator-0.1.2/table_validator.egg-info}/PKG-INFO +1 -1
  5. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/scm_version.json +2 -2
  6. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_catalog_validator.py +40 -0
  7. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_cli.py +30 -0
  8. {table_validator-0.1.0 → table_validator-0.1.2}/LICENSE +0 -0
  9. {table_validator-0.1.0 → table_validator-0.1.2}/README.md +0 -0
  10. {table_validator-0.1.0 → table_validator-0.1.2}/pyproject.toml +0 -0
  11. {table_validator-0.1.0 → table_validator-0.1.2}/setup.cfg +0 -0
  12. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/__init__.py +0 -0
  13. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/auth/__init__.py +0 -0
  14. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/auth/azure_auth.py +0 -0
  15. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/auth/databricks_auth.py +0 -0
  16. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/__init__.py +0 -0
  17. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/partition_prompt.py +0 -0
  18. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/summary_table.py +0 -0
  19. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/cli/wizard.py +0 -0
  20. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/config/__init__.py +0 -0
  21. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/config/manager.py +0 -0
  22. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/config/schema.py +0 -0
  23. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/__init__.py +0 -0
  24. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/azure_connector.py +0 -0
  25. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/connectors/databricks_connector.py +0 -0
  26. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/engine/__init__.py +0 -0
  27. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/engine/comparison_engine.py +0 -0
  28. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/models.py +0 -0
  29. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/reports/__init__.py +0 -0
  30. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/reports/excel_report.py +0 -0
  31. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/__init__.py +0 -0
  32. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/blob_discovery.py +0 -0
  33. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator/validators/row_validator.py +0 -0
  34. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/SOURCES.txt +0 -0
  35. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/dependency_links.txt +0 -0
  36. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/entry_points.txt +0 -0
  37. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/requires.txt +0 -0
  38. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/scm_file_list.json +0 -0
  39. {table_validator-0.1.0 → table_validator-0.1.2}/table_validator.egg-info/top_level.txt +0 -0
  40. {table_validator-0.1.0 → table_validator-0.1.2}/tests/__init__.py +0 -0
  41. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_blob_discovery.py +0 -0
  42. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_databricks_connector.py +0 -0
  43. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_excel_report.py +0 -0
  44. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_partition_prompt.py +0 -0
  45. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_report_command.py +0 -0
  46. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_row_validator.py +0 -0
  47. {table_validator-0.1.0 → table_validator-0.1.2}/tests/test_wizard.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.0
3
+ Version: 0.1.2
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -63,6 +63,14 @@ def _configure_logging(verbose: bool, quiet: bool) -> None:
63
63
  logger = logging.getLogger("table_validator")
64
64
  logger.propagate = False
65
65
 
66
+ # Each CLI invocation must start with a clean handler set: leftover
67
+ # handlers from a previous invocation in the same process (e.g. every
68
+ # test in this suite drives the CLI through the same interpreter) can
69
+ # point at an already-closed/detached stream, causing "Logging error"
70
+ # noise on emit rather than the intended output.
71
+ for old_handler in list(logger.handlers):
72
+ logger.removeHandler(old_handler)
73
+
66
74
  if quiet:
67
75
  # No handler at all (not even a raised level) - otherwise Python's
68
76
  # logging module falls back to its own stderr "lastResort" handler
@@ -124,6 +132,9 @@ def info() -> None:
124
132
  " Runs the comparison using the saved configuration and "
125
133
  "writes an Excel report (validation_report.xlsx by default). "
126
134
  "Prints a pass/fail summary in the terminal when it finishes.\n"
135
+ " To write to a different file instead of overwriting the "
136
+ "default (e.g. if it's still open in Excel from a previous run):\n"
137
+ " tablevalidator validate --output validation_report_new.xlsx\n"
127
138
  "\n"
128
139
  " 3. tablevalidator open\n"
129
140
  " Opens the most recently generated report in your default "
@@ -135,7 +146,9 @@ def info() -> None:
135
146
  "existing report without opening it.\n"
136
147
  "\n"
137
148
  "Run 'tablevalidator <command> --help' for a command's full "
138
- "options (e.g. tablevalidator validate --help)."
149
+ "options (e.g. tablevalidator validate --help).\n"
150
+ "\n"
151
+ "TO KEERTHIVASAN"
139
152
  )
140
153
 
141
154
 
@@ -292,11 +305,21 @@ def validate(
292
305
  if config.source_type == SourceType.DATABRICKS
293
306
  else None
294
307
  )
295
- generate_excel_report(
296
- result, str(output),
297
- source_type=config.source_type.value,
298
- enabled_validations=report_enabled_validations,
299
- )
308
+ try:
309
+ generate_excel_report(
310
+ result, str(output),
311
+ source_type=config.source_type.value,
312
+ enabled_validations=report_enabled_validations,
313
+ )
314
+ except PermissionError:
315
+ typer.secho(
316
+ f"Could not write '{output}' - the file is open in Excel (or "
317
+ "another program) and locked for writing. Close it and re-run "
318
+ "'tablevalidator validate', or pass a different path with "
319
+ "--output.",
320
+ fg=typer.colors.RED,
321
+ )
322
+ raise typer.Exit(code=1)
300
323
 
301
324
  # ------------------------------------------------------------------
302
325
  # 8. Print a summary - per-table PASS/FAIL lines, then the same
@@ -88,6 +88,30 @@ class CatalogValidator:
88
88
  self.partition_prompt = partition_prompt
89
89
  logger.debug("CatalogValidator initialised")
90
90
 
91
+ @staticmethod
92
+ def _lookup_primary_key(
93
+ request: "CatalogValidationRequest",
94
+ schema_name: str,
95
+ table_name: str,
96
+ ) -> Optional[List[str]]:
97
+ """
98
+ Look up request.primary_keys (keyed by "schema.table" or bare
99
+ table name, as typed by the user into config.yaml/the wizard) for
100
+ the given schema_name/table_name.
101
+
102
+ schema_name/table_name here come from whatever casing Databricks'
103
+ information_schema actually returns, matched case-insensitively
104
+ against the catalog in compare_schemas/compare_tables - which
105
+ does not necessarily match the exact casing the user configured.
106
+ An exact-match dict lookup would silently miss a configured key
107
+ and fall through to the much more expensive ROW_NUMBER() fallback
108
+ with no indication why, so this normalizes both sides to
109
+ lowercase before comparing.
110
+ """
111
+ lowered = {k.lower(): v for k, v in request.primary_keys.items()}
112
+ key_lookup = f"{schema_name}.{table_name}".lower()
113
+ return lowered.get(key_lookup) or lowered.get(table_name.lower())
114
+
91
115
  # ------------------------------------------------------------------
92
116
  # Stage 1 + top-level orchestration
93
117
  # ------------------------------------------------------------------
@@ -697,10 +721,7 @@ class CatalogValidator:
697
721
 
698
722
  # Configured PK column missing from either side is also BLOCKING -
699
723
  # every later tier depends on being able to resolve a usable key.
700
- key_lookup = f"{schema_name}.{table_name}"
701
- configured_key = request.primary_keys.get(key_lookup) or request.primary_keys.get(
702
- table_name
703
- )
724
+ configured_key = self._lookup_primary_key(request, schema_name, table_name)
704
725
  if configured_key:
705
726
  common_lower = {c.lower() for c in common_cols}
706
727
  if any(k.lower() not in common_lower for k in configured_key):
@@ -1040,8 +1061,7 @@ class CatalogValidator:
1040
1061
  )
1041
1062
  return
1042
1063
 
1043
- key_lookup = f"{schema_name}.{table_name}"
1044
- key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(table_name)
1064
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1045
1065
  candidates = self._partition_candidates(common_cols, key_columns)
1046
1066
 
1047
1067
  context = PartitionPromptContext(
@@ -1184,10 +1204,7 @@ class CatalogValidator:
1184
1204
  result.tier_reached = ValidationTier.ROW_HASH
1185
1205
 
1186
1206
  if not using_row_number_fallback:
1187
- key_lookup = f"{schema_name}.{table_name}"
1188
- key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
1189
- table_name
1190
- )
1207
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1191
1208
  mismatched_keys = [m.primary_key for m in mismatches if m.status == "MISMATCH"]
1192
1209
  if key_columns and mismatched_keys:
1193
1210
  self._tier5_column_diff(
@@ -1238,10 +1255,7 @@ class CatalogValidator:
1238
1255
  written straight to `result.data`, so multiple bucket calls
1239
1256
  aggregate instead of each overwriting the last.
1240
1257
  """
1241
- key_lookup = f"{schema_name}.{table_name}"
1242
- row_hash_key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
1243
- table_name
1244
- )
1258
+ row_hash_key_columns = self._lookup_primary_key(request, schema_name, table_name)
1245
1259
 
1246
1260
  using_row_number_fallback = not row_hash_key_columns
1247
1261
  if using_row_number_fallback:
@@ -1470,10 +1484,7 @@ class CatalogValidator:
1470
1484
  ) -> DataValidationResult:
1471
1485
 
1472
1486
  mode = request.data_compare_mode
1473
- key = f"{schema_name}.{table_name}"
1474
- key_columns = request.primary_keys.get(key) or request.primary_keys.get(
1475
- table_name
1476
- )
1487
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1477
1488
 
1478
1489
  logger.info(
1479
1490
  "[compare_data] table=%s.%s | mode=%s | resolved_key_columns=%s",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.0
3
+ Version: 0.1.2
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -1,7 +1,7 @@
1
1
  {
2
- "tag": "0.1.0",
2
+ "tag": "0.1.2",
3
3
  "distance": 0,
4
- "node": "gaeb6df763293e80acabcfa9ac9c4d5e4a2f2e8f1",
4
+ "node": "g43988f9f51a097e239c49384ab59f00fec5851ab",
5
5
  "dirty": false,
6
6
  "branch": "HEAD",
7
7
  "node_date": "2026-08-27"
@@ -991,6 +991,46 @@ def test_tier5_column_diff_runs_for_mismatched_key_and_names_exact_column():
991
991
  assert detail.verified is True
992
992
 
993
993
 
994
+ def test_primary_key_lookup_is_case_insensitive_against_catalog_metadata():
995
+ """request.primary_keys is keyed by whatever casing the user typed
996
+ into config.yaml/the wizard, but schema_name/table_name at lookup
997
+ time come from Databricks' information_schema (matched
998
+ case-insensitively against the catalog elsewhere in this file). A
999
+ primary key configured as "Bronze.Customers" must still be found for
1000
+ the real "bronze.customers" table - an exact-match lookup would
1001
+ silently miss it and fall through to the much slower ROW_NUMBER()
1002
+ fallback instead of using the real configured key."""
1003
+ connector = _mismatched_fingerprint_connector()
1004
+ connector.get_row_hashes.side_effect = lambda catalog, schema, table, cols, pk, bucket_predicate=None: (
1005
+ _hash_df([(1, "aaa"), (2, "bbb")])
1006
+ if catalog == "cat_source"
1007
+ else _hash_df([(1, "aaa"), (2, "zzz")])
1008
+ )
1009
+ connector.get_row_detail_for_keys.return_value = [
1010
+ {
1011
+ "key": {"id": 2},
1012
+ "mismatched_columns": ["name"],
1013
+ "source_values": {"name": "old-value"},
1014
+ "target_values": {"name": "new-value"},
1015
+ "source_row_hash": "bbb",
1016
+ "target_row_hash": "zzz",
1017
+ }
1018
+ ]
1019
+ validator = CatalogValidator(connector)
1020
+
1021
+ result = validator.compare_catalogs(
1022
+ _request(primary_keys={"Bronze.Customers": ["id"]})
1023
+ )
1024
+ table = result.schemas[0].tables[0]
1025
+
1026
+ connector.get_row_hashes_by_row_number.assert_not_called()
1027
+ assert connector.get_row_hashes.call_count == 2 # once per side (source, target)
1028
+ connector.get_row_detail_for_keys.assert_called_once()
1029
+ assert table.data.key_columns == ["id"]
1030
+ assert table.tier_reached == ValidationTier.COLUMN_DIFF
1031
+ assert table.data.sample_changed_detail[0].verified is True
1032
+
1033
+
994
1034
  def test_tier5_column_diff_runs_for_row_number_fallback_when_mismatched():
995
1035
  """No primary key configured -> row-number fallback is used for Tier
996
1036
  4, but a real mismatch must still let Tier 5 attempt best-effort
@@ -297,6 +297,36 @@ def test_open_command_errors_cleanly_when_report_missing(tmp_path: Path, _never_
297
297
  _never_launch_a_real_app.assert_not_called()
298
298
 
299
299
 
300
+ def test_validate_locked_report_file_errors_cleanly_not_traceback(tmp_path: Path) -> None:
301
+ config_path = _full_config(tmp_path)
302
+ output_path = tmp_path / "validation_report.xlsx"
303
+ mock_connector = _mock_databricks_connector()
304
+
305
+ with patch(
306
+ "table_validator.cli.main.DatabricksConnector", return_value=mock_connector
307
+ ), patch(
308
+ "table_validator.cli.main.get_databricks_token", return_value="dapi_fake"
309
+ ), patch(
310
+ "table_validator.cli.main.get_azure_credential", return_value=None
311
+ ), patch(
312
+ "table_validator.cli.main.generate_excel_report",
313
+ side_effect=PermissionError(13, "Permission denied"),
314
+ ):
315
+ result = runner.invoke(
316
+ app,
317
+ [
318
+ "validate",
319
+ "--config-path", str(config_path),
320
+ "--output", str(output_path),
321
+ ],
322
+ )
323
+
324
+ assert result.exit_code == 1
325
+ assert "open in Excel" in result.output
326
+ assert "--output" in result.output
327
+ assert "Traceback" not in result.output
328
+
329
+
300
330
  def test_validate_exits_nonzero_on_failed_validation(tmp_path: Path) -> None:
301
331
  config_path = _full_config(tmp_path)
302
332
  output_path = tmp_path / "validation_report.xlsx"
File without changes