table-validator 0.1.1__tar.gz → 0.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {table_validator-0.1.1/table_validator.egg-info → table_validator-0.1.3}/PKG-INFO +9 -1
  2. {table_validator-0.1.1 → table_validator-0.1.3}/README.md +8 -0
  3. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/connectors/databricks_connector.py +29 -5
  4. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/validators/catalog_validator.py +29 -18
  5. {table_validator-0.1.1 → table_validator-0.1.3/table_validator.egg-info}/PKG-INFO +9 -1
  6. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator.egg-info/scm_version.json +2 -2
  7. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_catalog_validator.py +40 -0
  8. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_databricks_connector.py +78 -0
  9. {table_validator-0.1.1 → table_validator-0.1.3}/LICENSE +0 -0
  10. {table_validator-0.1.1 → table_validator-0.1.3}/pyproject.toml +0 -0
  11. {table_validator-0.1.1 → table_validator-0.1.3}/setup.cfg +0 -0
  12. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/__init__.py +0 -0
  13. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/auth/__init__.py +0 -0
  14. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/auth/azure_auth.py +0 -0
  15. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/auth/databricks_auth.py +0 -0
  16. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/cli/__init__.py +0 -0
  17. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/cli/main.py +0 -0
  18. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/cli/partition_prompt.py +0 -0
  19. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/cli/summary_table.py +0 -0
  20. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/cli/wizard.py +0 -0
  21. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/config/__init__.py +0 -0
  22. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/config/manager.py +0 -0
  23. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/config/schema.py +0 -0
  24. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/connectors/__init__.py +0 -0
  25. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/connectors/azure_connector.py +0 -0
  26. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/engine/__init__.py +0 -0
  27. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/engine/comparison_engine.py +0 -0
  28. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/models.py +0 -0
  29. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/reports/__init__.py +0 -0
  30. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/reports/excel_report.py +0 -0
  31. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/validators/__init__.py +0 -0
  32. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/validators/blob_discovery.py +0 -0
  33. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator/validators/row_validator.py +0 -0
  34. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator.egg-info/SOURCES.txt +0 -0
  35. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator.egg-info/dependency_links.txt +0 -0
  36. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator.egg-info/entry_points.txt +0 -0
  37. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator.egg-info/requires.txt +0 -0
  38. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator.egg-info/scm_file_list.json +0 -0
  39. {table_validator-0.1.1 → table_validator-0.1.3}/table_validator.egg-info/top_level.txt +0 -0
  40. {table_validator-0.1.1 → table_validator-0.1.3}/tests/__init__.py +0 -0
  41. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_blob_discovery.py +0 -0
  42. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_cli.py +0 -0
  43. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_excel_report.py +0 -0
  44. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_partition_prompt.py +0 -0
  45. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_report_command.py +0 -0
  46. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_row_validator.py +0 -0
  47. {table_validator-0.1.1 → table_validator-0.1.3}/tests/test_wizard.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.1
3
+ Version: 0.1.3
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -174,6 +174,14 @@ Everything lives outside the repo, under your home directory:
174
174
  by construction, since it's written under your home directory rather
175
175
  than the working directory.
176
176
 
177
+ Optional: `DATABRICKS_RETRY_TIMEOUT_SECONDS` (a plain environment variable,
178
+ not part of `config.yaml`) raises the CloudFetch HTTP retry timeout above
179
+ its 300-second default. On a slow or unstable network, downloading a
180
+ large row-hash result set can legitimately take longer than that, causing
181
+ a `Retry request would exceed Retry policy max retry duration` failure
182
+ even though the query itself succeeded. Set it higher (e.g. `900` for 15
183
+ minutes) if you hit this on large tables.
184
+
177
185
  A future version will replace manual credential entry with Azure CLI /
178
186
  Service Principal auth and Databricks CLI / OAuth login, without changing
179
187
  the config file format or any command usage above.
@@ -147,6 +147,14 @@ Everything lives outside the repo, under your home directory:
147
147
  by construction, since it's written under your home directory rather
148
148
  than the working directory.
149
149
 
150
+ Optional: `DATABRICKS_RETRY_TIMEOUT_SECONDS` (a plain environment variable,
151
+ not part of `config.yaml`) raises the CloudFetch HTTP retry timeout above
152
+ its 300-second default. On a slow or unstable network, downloading a
153
+ large row-hash result set can legitimately take longer than that, causing
154
+ a `Retry request would exceed Retry policy max retry duration` failure
155
+ even though the query itself succeeded. Set it higher (e.g. `900` for 15
156
+ minutes) if you hit this on large tables.
157
+
150
158
  A future version will replace manual credential entry with Azure CLI /
151
159
  Service Principal auth and Databricks CLI / OAuth login, without changing
152
160
  the config file format or any command usage above.
@@ -14,6 +14,7 @@ from __future__ import annotations
14
14
  import datetime
15
15
  import logging
16
16
  import numbers
17
+ import os
17
18
  from typing import TYPE_CHECKING, Any, Dict, List, Optional, Sequence, Tuple
18
19
 
19
20
  import pandas as pd
@@ -97,6 +98,7 @@ class DatabricksConnector:
97
98
  host: Optional[str] = None,
98
99
  token: Optional[str] = None,
99
100
  http_path: Optional[str] = None,
101
+ retry_timeout_seconds: Optional[float] = None,
100
102
  ) -> None:
101
103
  """
102
104
  host/token/http_path must be resolved by the caller before
@@ -104,11 +106,27 @@ class DatabricksConnector:
104
106
  get_databricks_token() for the token, and config.databricks.
105
107
  workspace_url/http_path for the rest. This connector does not
106
108
  read credentials from the environment itself.
109
+
110
+ retry_timeout_seconds overrides databricks-sql-connector's
111
+ CloudFetch HTTP retry policy (_retry_stop_after_attempts_duration),
112
+ which otherwise defaults to 300 seconds. On a slow or unstable
113
+ network, downloading a large row-hash result set can legitimately
114
+ take longer than that, causing a hard
115
+ "Retry request would exceed Retry policy max retry duration"
116
+ failure even though the query itself succeeded - raising this
117
+ gives a slow-but-working connection more time instead of giving
118
+ up. Falls back to the DATABRICKS_RETRY_TIMEOUT_SECONDS environment
119
+ variable, then the connector's own default, if not given.
107
120
  """
108
121
 
109
122
  self._host = host
110
123
  self._token = token
111
124
  self._http_path = http_path
125
+ self._retry_timeout_seconds = retry_timeout_seconds or (
126
+ float(os.environ["DATABRICKS_RETRY_TIMEOUT_SECONDS"])
127
+ if os.environ.get("DATABRICKS_RETRY_TIMEOUT_SECONDS")
128
+ else None
129
+ )
112
130
 
113
131
  if not self._host or not self._token:
114
132
  raise ValueError(
@@ -138,11 +156,17 @@ class DatabricksConnector:
138
156
 
139
157
  try:
140
158
 
141
- self._connection = sql.connect(
142
- server_hostname=self._host,
143
- http_path=self._http_path,
144
- access_token=self._token,
145
- )
159
+ connect_kwargs: Dict[str, Any] = {
160
+ "server_hostname": self._host,
161
+ "http_path": self._http_path,
162
+ "access_token": self._token,
163
+ }
164
+ if self._retry_timeout_seconds is not None:
165
+ connect_kwargs["_retry_stop_after_attempts_duration"] = (
166
+ self._retry_timeout_seconds
167
+ )
168
+
169
+ self._connection = sql.connect(**connect_kwargs)
146
170
 
147
171
  with self._connection.cursor() as cursor:
148
172
  cursor.execute("SELECT 1")
@@ -88,6 +88,30 @@ class CatalogValidator:
88
88
  self.partition_prompt = partition_prompt
89
89
  logger.debug("CatalogValidator initialised")
90
90
 
91
+ @staticmethod
92
+ def _lookup_primary_key(
93
+ request: "CatalogValidationRequest",
94
+ schema_name: str,
95
+ table_name: str,
96
+ ) -> Optional[List[str]]:
97
+ """
98
+ Look up request.primary_keys (keyed by "schema.table" or bare
99
+ table name, as typed by the user into config.yaml/the wizard) for
100
+ the given schema_name/table_name.
101
+
102
+ schema_name/table_name here come from whatever casing Databricks'
103
+ information_schema actually returns, matched case-insensitively
104
+ against the catalog in compare_schemas/compare_tables - which
105
+ does not necessarily match the exact casing the user configured.
106
+ An exact-match dict lookup would silently miss a configured key
107
+ and fall through to the much more expensive ROW_NUMBER() fallback
108
+ with no indication why, so this normalizes both sides to
109
+ lowercase before comparing.
110
+ """
111
+ lowered = {k.lower(): v for k, v in request.primary_keys.items()}
112
+ key_lookup = f"{schema_name}.{table_name}".lower()
113
+ return lowered.get(key_lookup) or lowered.get(table_name.lower())
114
+
91
115
  # ------------------------------------------------------------------
92
116
  # Stage 1 + top-level orchestration
93
117
  # ------------------------------------------------------------------
@@ -697,10 +721,7 @@ class CatalogValidator:
697
721
 
698
722
  # Configured PK column missing from either side is also BLOCKING -
699
723
  # every later tier depends on being able to resolve a usable key.
700
- key_lookup = f"{schema_name}.{table_name}"
701
- configured_key = request.primary_keys.get(key_lookup) or request.primary_keys.get(
702
- table_name
703
- )
724
+ configured_key = self._lookup_primary_key(request, schema_name, table_name)
704
725
  if configured_key:
705
726
  common_lower = {c.lower() for c in common_cols}
706
727
  if any(k.lower() not in common_lower for k in configured_key):
@@ -1040,8 +1061,7 @@ class CatalogValidator:
1040
1061
  )
1041
1062
  return
1042
1063
 
1043
- key_lookup = f"{schema_name}.{table_name}"
1044
- key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(table_name)
1064
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1045
1065
  candidates = self._partition_candidates(common_cols, key_columns)
1046
1066
 
1047
1067
  context = PartitionPromptContext(
@@ -1184,10 +1204,7 @@ class CatalogValidator:
1184
1204
  result.tier_reached = ValidationTier.ROW_HASH
1185
1205
 
1186
1206
  if not using_row_number_fallback:
1187
- key_lookup = f"{schema_name}.{table_name}"
1188
- key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
1189
- table_name
1190
- )
1207
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1191
1208
  mismatched_keys = [m.primary_key for m in mismatches if m.status == "MISMATCH"]
1192
1209
  if key_columns and mismatched_keys:
1193
1210
  self._tier5_column_diff(
@@ -1238,10 +1255,7 @@ class CatalogValidator:
1238
1255
  written straight to `result.data`, so multiple bucket calls
1239
1256
  aggregate instead of each overwriting the last.
1240
1257
  """
1241
- key_lookup = f"{schema_name}.{table_name}"
1242
- row_hash_key_columns = request.primary_keys.get(key_lookup) or request.primary_keys.get(
1243
- table_name
1244
- )
1258
+ row_hash_key_columns = self._lookup_primary_key(request, schema_name, table_name)
1245
1259
 
1246
1260
  using_row_number_fallback = not row_hash_key_columns
1247
1261
  if using_row_number_fallback:
@@ -1470,10 +1484,7 @@ class CatalogValidator:
1470
1484
  ) -> DataValidationResult:
1471
1485
 
1472
1486
  mode = request.data_compare_mode
1473
- key = f"{schema_name}.{table_name}"
1474
- key_columns = request.primary_keys.get(key) or request.primary_keys.get(
1475
- table_name
1476
- )
1487
+ key_columns = self._lookup_primary_key(request, schema_name, table_name)
1477
1488
 
1478
1489
  logger.info(
1479
1490
  "[compare_data] table=%s.%s | mode=%s | resolved_key_columns=%s",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.1
3
+ Version: 0.1.3
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -174,6 +174,14 @@ Everything lives outside the repo, under your home directory:
174
174
  by construction, since it's written under your home directory rather
175
175
  than the working directory.
176
176
 
177
+ Optional: `DATABRICKS_RETRY_TIMEOUT_SECONDS` (a plain environment variable,
178
+ not part of `config.yaml`) raises the CloudFetch HTTP retry timeout above
179
+ its 300-second default. On a slow or unstable network, downloading a
180
+ large row-hash result set can legitimately take longer than that, causing
181
+ a `Retry request would exceed Retry policy max retry duration` failure
182
+ even though the query itself succeeded. Set it higher (e.g. `900` for 15
183
+ minutes) if you hit this on large tables.
184
+
177
185
  A future version will replace manual credential entry with Azure CLI /
178
186
  Service Principal auth and Databricks CLI / OAuth login, without changing
179
187
  the config file format or any command usage above.
@@ -1,7 +1,7 @@
1
1
  {
2
- "tag": "0.1.1",
2
+ "tag": "0.1.3",
3
3
  "distance": 0,
4
- "node": "gd0c901c9dfc750527e33620c5dea6b15e43e4443",
4
+ "node": "g8f7a326158b935508f39bc9369245d7817f1dc89",
5
5
  "dirty": false,
6
6
  "branch": "HEAD",
7
7
  "node_date": "2026-08-27"
@@ -991,6 +991,46 @@ def test_tier5_column_diff_runs_for_mismatched_key_and_names_exact_column():
991
991
  assert detail.verified is True
992
992
 
993
993
 
994
+ def test_primary_key_lookup_is_case_insensitive_against_catalog_metadata():
995
+ """request.primary_keys is keyed by whatever casing the user typed
996
+ into config.yaml/the wizard, but schema_name/table_name at lookup
997
+ time come from Databricks' information_schema (matched
998
+ case-insensitively against the catalog elsewhere in this file). A
999
+ primary key configured as "Bronze.Customers" must still be found for
1000
+ the real "bronze.customers" table - an exact-match lookup would
1001
+ silently miss it and fall through to the much slower ROW_NUMBER()
1002
+ fallback instead of using the real configured key."""
1003
+ connector = _mismatched_fingerprint_connector()
1004
+ connector.get_row_hashes.side_effect = lambda catalog, schema, table, cols, pk, bucket_predicate=None: (
1005
+ _hash_df([(1, "aaa"), (2, "bbb")])
1006
+ if catalog == "cat_source"
1007
+ else _hash_df([(1, "aaa"), (2, "zzz")])
1008
+ )
1009
+ connector.get_row_detail_for_keys.return_value = [
1010
+ {
1011
+ "key": {"id": 2},
1012
+ "mismatched_columns": ["name"],
1013
+ "source_values": {"name": "old-value"},
1014
+ "target_values": {"name": "new-value"},
1015
+ "source_row_hash": "bbb",
1016
+ "target_row_hash": "zzz",
1017
+ }
1018
+ ]
1019
+ validator = CatalogValidator(connector)
1020
+
1021
+ result = validator.compare_catalogs(
1022
+ _request(primary_keys={"Bronze.Customers": ["id"]})
1023
+ )
1024
+ table = result.schemas[0].tables[0]
1025
+
1026
+ connector.get_row_hashes_by_row_number.assert_not_called()
1027
+ assert connector.get_row_hashes.call_count == 2 # once per side (source, target)
1028
+ connector.get_row_detail_for_keys.assert_called_once()
1029
+ assert table.data.key_columns == ["id"]
1030
+ assert table.tier_reached == ValidationTier.COLUMN_DIFF
1031
+ assert table.data.sample_changed_detail[0].verified is True
1032
+
1033
+
994
1034
  def test_tier5_column_diff_runs_for_row_number_fallback_when_mismatched():
995
1035
  """No primary key configured -> row-number fallback is used for Tier
996
1036
  4, but a real mismatch must still let Tier 5 attempt best-effort
@@ -8,9 +8,12 @@ so these tests verify SQL-generation logic and result parsing only.
8
8
 
9
9
  from __future__ import annotations
10
10
 
11
+ from unittest.mock import MagicMock
12
+
11
13
  import pandas as pd
12
14
  import pytest
13
15
 
16
+ import table_validator.connectors.databricks_connector as databricks_connector_module
14
17
  from table_validator.connectors.databricks_connector import DatabricksConnector
15
18
 
16
19
 
@@ -402,3 +405,78 @@ def test_get_row_detail_for_row_numbers_wraps_exceptions(monkeypatch):
402
405
  row_numbers=[2],
403
406
  value_columns=["id"],
404
407
  )
408
+
409
+
410
+ # ---------------------------------------------------------------------------
411
+ # connect(): retry timeout override for slow/unstable CloudFetch downloads.
412
+ # ---------------------------------------------------------------------------
413
+ def test_connect_omits_retry_kwarg_by_default(monkeypatch):
414
+ """Without an explicit retry_timeout_seconds or
415
+ DATABRICKS_RETRY_TIMEOUT_SECONDS, connect() must not pass
416
+ _retry_stop_after_attempts_duration at all, preserving
417
+ databricks-sql-connector's own default (300s)."""
418
+ monkeypatch.delenv("DATABRICKS_RETRY_TIMEOUT_SECONDS", raising=False)
419
+ captured = {}
420
+
421
+ def fake_connect(**kwargs):
422
+ captured.update(kwargs)
423
+ return MagicMock()
424
+
425
+ monkeypatch.setattr(databricks_connector_module.sql, "connect", fake_connect)
426
+ connector = _connector()
427
+
428
+ connector.connect()
429
+
430
+ assert "_retry_stop_after_attempts_duration" not in captured
431
+
432
+
433
+ def test_connect_passes_explicit_retry_timeout_seconds(monkeypatch):
434
+ monkeypatch.delenv("DATABRICKS_RETRY_TIMEOUT_SECONDS", raising=False)
435
+ captured = {}
436
+
437
+ def fake_connect(**kwargs):
438
+ captured.update(kwargs)
439
+ return MagicMock()
440
+
441
+ monkeypatch.setattr(databricks_connector_module.sql, "connect", fake_connect)
442
+ connector = DatabricksConnector(
443
+ host="h", token="t", http_path="p", retry_timeout_seconds=900.0
444
+ )
445
+
446
+ connector.connect()
447
+
448
+ assert captured["_retry_stop_after_attempts_duration"] == 900.0
449
+
450
+
451
+ def test_connect_falls_back_to_retry_timeout_env_var(monkeypatch):
452
+ monkeypatch.setenv("DATABRICKS_RETRY_TIMEOUT_SECONDS", "1200")
453
+ captured = {}
454
+
455
+ def fake_connect(**kwargs):
456
+ captured.update(kwargs)
457
+ return MagicMock()
458
+
459
+ monkeypatch.setattr(databricks_connector_module.sql, "connect", fake_connect)
460
+ connector = _connector()
461
+
462
+ connector.connect()
463
+
464
+ assert captured["_retry_stop_after_attempts_duration"] == 1200.0
465
+
466
+
467
+ def test_connect_explicit_arg_takes_priority_over_env_var(monkeypatch):
468
+ monkeypatch.setenv("DATABRICKS_RETRY_TIMEOUT_SECONDS", "1200")
469
+ captured = {}
470
+
471
+ def fake_connect(**kwargs):
472
+ captured.update(kwargs)
473
+ return MagicMock()
474
+
475
+ monkeypatch.setattr(databricks_connector_module.sql, "connect", fake_connect)
476
+ connector = DatabricksConnector(
477
+ host="h", token="t", http_path="p", retry_timeout_seconds=600.0
478
+ )
479
+
480
+ connector.connect()
481
+
482
+ assert captured["_retry_stop_after_attempts_duration"] == 600.0
File without changes