table-validator 0.1.6__tar.gz → 0.1.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {table_validator-0.1.6/table_validator.egg-info → table_validator-0.1.7}/PKG-INFO +1 -1
  2. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/cli/main.py +4 -1
  3. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/config/schema.py +36 -0
  4. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/models.py +33 -1
  5. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/validators/catalog_validator.py +28 -2
  6. {table_validator-0.1.6 → table_validator-0.1.7/table_validator.egg-info}/PKG-INFO +1 -1
  7. table_validator-0.1.7/table_validator.egg-info/scm_version.json +8 -0
  8. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_catalog_validator.py +111 -0
  9. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_cli.py +45 -0
  10. table_validator-0.1.6/table_validator.egg-info/scm_version.json +0 -8
  11. {table_validator-0.1.6 → table_validator-0.1.7}/LICENSE +0 -0
  12. {table_validator-0.1.6 → table_validator-0.1.7}/README.md +0 -0
  13. {table_validator-0.1.6 → table_validator-0.1.7}/pyproject.toml +0 -0
  14. {table_validator-0.1.6 → table_validator-0.1.7}/setup.cfg +0 -0
  15. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/__init__.py +0 -0
  16. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/auth/__init__.py +0 -0
  17. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/auth/azure_auth.py +0 -0
  18. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/auth/databricks_auth.py +0 -0
  19. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/cli/__init__.py +0 -0
  20. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/cli/partition_prompt.py +0 -0
  21. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/cli/summary_table.py +0 -0
  22. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/cli/wizard.py +0 -0
  23. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/config/__init__.py +0 -0
  24. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/config/manager.py +0 -0
  25. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/connectors/__init__.py +0 -0
  26. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/connectors/azure_connector.py +0 -0
  27. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/connectors/databricks_connector.py +0 -0
  28. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/engine/__init__.py +0 -0
  29. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/engine/comparison_engine.py +0 -0
  30. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/reports/__init__.py +0 -0
  31. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/reports/excel_report.py +0 -0
  32. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/validators/__init__.py +0 -0
  33. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/validators/blob_discovery.py +0 -0
  34. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator/validators/row_validator.py +0 -0
  35. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator.egg-info/SOURCES.txt +0 -0
  36. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator.egg-info/dependency_links.txt +0 -0
  37. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator.egg-info/entry_points.txt +0 -0
  38. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator.egg-info/requires.txt +0 -0
  39. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator.egg-info/scm_file_list.json +0 -0
  40. {table_validator-0.1.6 → table_validator-0.1.7}/table_validator.egg-info/top_level.txt +0 -0
  41. {table_validator-0.1.6 → table_validator-0.1.7}/tests/__init__.py +0 -0
  42. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_blob_discovery.py +0 -0
  43. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_databricks_connector.py +0 -0
  44. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_excel_report.py +0 -0
  45. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_partition_prompt.py +0 -0
  46. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_report_command.py +0 -0
  47. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_row_validator.py +0 -0
  48. {table_validator-0.1.6 → table_validator-0.1.7}/tests/test_wizard.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.6
3
+ Version: 0.1.7
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -148,7 +148,7 @@ def info() -> None:
148
148
  "Run 'tablevalidator <command> --help' for a command's full "
149
149
  "options (e.g. tablevalidator validate --help).\n"
150
150
  "\n"
151
- "TO KEERTHIVASAN"
151
+ "-MS-"
152
152
  )
153
153
 
154
154
 
@@ -470,6 +470,9 @@ def _run_databricks_validation(
470
470
  enabled_validations=set(config.validations),
471
471
  primary_keys=primary_keys,
472
472
  max_tier=max_tier,
473
+ only_columns=config.only_columns,
474
+ ignore_columns=config.ignore_columns,
475
+ ignore_datatype_columns=config.ignore_datatype_columns,
473
476
  )
474
477
 
475
478
  partition_prompt = build_partition_prompt(yes=yes)
@@ -166,6 +166,42 @@ class ValidatorConfig(BaseModel):
166
166
  ),
167
167
  )
168
168
 
169
+ only_columns: Optional[List[str]] = Field(
170
+ default=None,
171
+ description=(
172
+ "Optional allowlist: if set, only these columns (plus the "
173
+ "primary key, if any) are compared - every other common "
174
+ "column is excluded from every check. Only meaningful for the "
175
+ "single named table in source_table/target_table, same scope "
176
+ "as primary_key. If a column named here isn't actually common "
177
+ "to both sides, it's silently absent from the effective set "
178
+ "(same convention as tables/schemas restrictions elsewhere). "
179
+ "If both only_columns and ignore_columns name the same "
180
+ "column, ignore_columns wins - it is always excluded."
181
+ ),
182
+ )
183
+
184
+ ignore_columns: List[str] = Field(
185
+ default_factory=list,
186
+ description=(
187
+ "Columns to exclude entirely from every check (name, type, "
188
+ "nullable, statistics, row-hash) - useful for auto-generated "
189
+ "columns like timestamps that are expected to always differ."
190
+ ),
191
+ )
192
+
193
+ ignore_datatype_columns: List[str] = Field(
194
+ default_factory=list,
195
+ description=(
196
+ "Columns whose data-type mismatch should be ignored - the "
197
+ "column's other checks (nullable, statistics, row-hash) still "
198
+ "run normally, but a real type difference here is reported as "
199
+ "SKIPPED rather than FAIL and never fails the table or aborts "
200
+ "the schema stage, even for a cross-family type change that "
201
+ "would otherwise be BLOCKING."
202
+ ),
203
+ )
204
+
169
205
  blob_source: BlobSourceConfig = Field(default_factory=BlobSourceConfig)
170
206
  sql_source: SqlSourceConfig = Field(default_factory=SqlSourceConfig)
171
207
 
@@ -488,6 +488,35 @@ class CatalogValidationRequest(BaseModel):
488
488
 
489
489
  ignore_columns: List[str] = Field(default_factory=list)
490
490
 
491
+ only_columns: Optional[List[str]] = Field(
492
+ default=None,
493
+ description=(
494
+ "Optional allowlist: if set, common_cols is further restricted "
495
+ "to just these column names (case-insensitive) before any "
496
+ "check runs - every other common column is excluded from name/"
497
+ "type/nullable/statistics/row-hash checks entirely, the same "
498
+ "as ignore_columns' exclusion. A name here that isn't actually "
499
+ "a common column is silently absent from the effective set, "
500
+ "matching this class's other restriction fields (schemas/ "
501
+ "tables). If a column appears in both only_columns and "
502
+ "ignore_columns, ignore_columns wins - it is always excluded, "
503
+ "applied after the allowlist restriction."
504
+ ),
505
+ )
506
+
507
+ ignore_datatype_columns: List[str] = Field(
508
+ default_factory=list,
509
+ description=(
510
+ "Columns (case-insensitive) whose data_type_status should be "
511
+ "reported as SKIPPED rather than PASS/FAIL, and excluded from "
512
+ "Tier 0's BLOCKING cross-family-type-change classification - "
513
+ "a real type difference on one of these columns never fails "
514
+ "the table or aborts the schema stage. The column's other "
515
+ "checks (nullable, null/distinct/min-max statistics, row-"
516
+ "hash) still run normally."
517
+ ),
518
+ )
519
+
491
520
  case_sensitive_columns: bool = Field(
492
521
  default=False,
493
522
  description="Case-sensitive column name comparison.",
@@ -569,7 +598,10 @@ class CatalogValidationRequest(BaseModel):
569
598
  ),
570
599
  )
571
600
 
572
- @field_validator("schemas", "tables", "ignore_columns", mode="before")
601
+ @field_validator(
602
+ "schemas", "tables", "ignore_columns", "only_columns",
603
+ "ignore_datatype_columns", mode="before",
604
+ )
573
605
  @classmethod
574
606
  def _ensure_list(cls, value: Any) -> Any:
575
607
  if value is None:
@@ -986,6 +986,19 @@ class CatalogValidator:
986
986
  source_schema_df, target_schema_df, request.case_sensitive_columns, ignore
987
987
  )
988
988
 
989
+ # only_columns (allowlist): restrict the already-computed common
990
+ # set to just these names, same case-insensitive convention as
991
+ # ignore_columns above. A name here that isn't actually a common
992
+ # column is silently absent from the effective set (matching
993
+ # tables/schemas' existing restriction-field convention) - it is
994
+ # NOT reported as missing/extra, since it was never confirmed to
995
+ # exist on both sides in the first place.
996
+ if request.only_columns:
997
+ allowed = {c.lower() for c in request.only_columns}
998
+ common_cols = [c for c in common_cols if c.lower() in allowed]
999
+
1000
+ ignore_datatype = {c.lower() for c in (request.ignore_datatype_columns or [])}
1001
+
989
1002
  if column_enabled:
990
1003
  result.missing_columns = missing_cols
991
1004
  result.extra_columns = extra_cols
@@ -1049,7 +1062,15 @@ class CatalogValidator:
1049
1062
  tgt_type = str(tgt_row.get("data_type"))
1050
1063
  col_result.source_data_type = src_type
1051
1064
  col_result.target_data_type = tgt_type
1052
- col_result.data_type_status = self._classify_type_family(src_type, tgt_type)
1065
+ if key in ignore_datatype:
1066
+ # User asked to ignore this column's datatype
1067
+ # entirely - report SKIPPED, not PASS/FAIL, so a real
1068
+ # type difference here is visibly disclosed as
1069
+ # "intentionally not checked" rather than looking
1070
+ # like the types happened to match.
1071
+ col_result.data_type_status = ValidationStatus.SKIPPED
1072
+ else:
1073
+ col_result.data_type_status = self._classify_type_family(src_type, tgt_type)
1053
1074
  dtype_statuses.append(col_result.data_type_status)
1054
1075
 
1055
1076
  if request.validate_nullable:
@@ -1076,9 +1097,14 @@ class CatalogValidator:
1076
1097
  result.nullable_status = ValidationStatus.SKIPPED
1077
1098
 
1078
1099
  # Cross-family type change is BLOCKING even when COLUMN reporting
1079
- # is disabled (detection always runs; only reporting is gated).
1100
+ # is disabled (detection always runs; only reporting is gated) -
1101
+ # except for a column the user explicitly put in
1102
+ # ignore_datatype_columns, which must never abort the table on a
1103
+ # type difference it was told to disregard.
1080
1104
  for col in common_cols:
1081
1105
  key = col.lower()
1106
+ if key in ignore_datatype:
1107
+ continue
1082
1108
  src_type = str(src_by_col.get(key, {}).get("data_type"))
1083
1109
  tgt_type = str(tgt_by_col.get(key, {}).get("data_type"))
1084
1110
  if self._classify_type_family(src_type, tgt_type) == ValidationStatus.FAIL:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.6
3
+ Version: 0.1.7
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -0,0 +1,8 @@
1
+ {
2
+ "tag": "0.1.7",
3
+ "distance": 0,
4
+ "node": "g989192ab5b13d03c174b2c33883a8569f5437e2a",
5
+ "dirty": false,
6
+ "branch": "HEAD",
7
+ "node_date": "2026-09-02"
8
+ }
@@ -1652,3 +1652,114 @@ def test_table_map_entry_matching_identical_name_is_not_duplicated():
1652
1652
 
1653
1653
  schema_result = result.schemas[0]
1654
1654
  assert len(schema_result.tables) == 1
1655
+
1656
+
1657
+ # ---------------------------------------------------------------------------
1658
+ # only_columns (allowlist) / ignore_datatype_columns
1659
+ # ---------------------------------------------------------------------------
1660
+ def test_only_columns_restricts_common_cols_to_allowlist():
1661
+ """only_columns further restricts the already-computed common set -
1662
+ a column not in the allowlist is excluded from every check (name/
1663
+ type/nullable/statistics), even though it's genuinely common to
1664
+ both sides."""
1665
+ connector = _make_connector()
1666
+ connector.get_table_schema.return_value = _schema_df(
1667
+ [("id", "int", False), ("name", "string", True), ("email", "string", True)]
1668
+ )
1669
+ connector.get_column_statistics.return_value = {
1670
+ "id": {"null_count": 0, "distinct_count": 100, "min": None, "max": None},
1671
+ "name": {"null_count": 2, "distinct_count": 95, "min": None, "max": None},
1672
+ "email": {"null_count": 0, "distinct_count": 100, "min": None, "max": None},
1673
+ }
1674
+ validator = CatalogValidator(connector)
1675
+
1676
+ result = validator.compare_catalogs(_request(only_columns=["id", "name"]))
1677
+
1678
+ table = result.schemas[0].tables[0]
1679
+ assert [c.column for c in table.columns] == ["id", "name"]
1680
+ assert "email" not in [c.column for c in table.columns]
1681
+ # Not reported as missing/extra - it was never confirmed absent, just
1682
+ # excluded by the allowlist.
1683
+ assert table.missing_columns == []
1684
+ assert table.extra_columns == []
1685
+
1686
+
1687
+ def test_only_columns_name_not_actually_common_is_silently_absent():
1688
+ """A name in only_columns that isn't a real common column doesn't
1689
+ error or appear anywhere - it's just absent from the effective set,
1690
+ matching tables/schemas' existing restriction-field convention."""
1691
+ connector = _make_connector()
1692
+ validator = CatalogValidator(connector)
1693
+
1694
+ result = validator.compare_catalogs(
1695
+ _request(only_columns=["id", "does_not_exist_column"])
1696
+ )
1697
+
1698
+ table = result.schemas[0].tables[0]
1699
+ assert [c.column for c in table.columns] == ["id"]
1700
+ assert result.status != ValidationStatus.ERROR
1701
+
1702
+
1703
+ def test_ignore_columns_wins_over_only_columns_for_same_column():
1704
+ """If a column appears in both only_columns and ignore_columns,
1705
+ ignore_columns wins - it is always excluded."""
1706
+ connector = _make_connector()
1707
+ validator = CatalogValidator(connector)
1708
+
1709
+ result = validator.compare_catalogs(
1710
+ _request(only_columns=["id", "name"], ignore_columns=["name"])
1711
+ )
1712
+
1713
+ table = result.schemas[0].tables[0]
1714
+ assert [c.column for c in table.columns] == ["id"]
1715
+
1716
+
1717
+ def test_ignore_datatype_columns_reports_skipped_not_fail():
1718
+ """A column in ignore_datatype_columns with genuinely different types
1719
+ across a type family (e.g. string vs int - normally BLOCKING) must
1720
+ report SKIPPED, never FAIL, and must not abort the table."""
1721
+ connector = _make_connector()
1722
+ connector.get_table_schema.side_effect = lambda catalog, schema, table: (
1723
+ _schema_df([("id", "int", False), ("legacy_flag", "string", True)])
1724
+ if catalog == "cat_source"
1725
+ else _schema_df([("id", "int", False), ("legacy_flag", "int", True)])
1726
+ )
1727
+ connector.get_column_statistics.return_value = {
1728
+ "id": {"null_count": 0, "distinct_count": 100, "min": None, "max": None},
1729
+ "legacy_flag": {"null_count": 0, "distinct_count": 2, "min": None, "max": None},
1730
+ }
1731
+ validator = CatalogValidator(connector)
1732
+
1733
+ result = validator.compare_catalogs(
1734
+ _request(ignore_datatype_columns=["legacy_flag"])
1735
+ )
1736
+
1737
+ table = result.schemas[0].tables[0]
1738
+ # Never BLOCKING - Tier 1+ actually ran (not SCHEMA_BLOCKED).
1739
+ assert table.tier_reached != ValidationTier.SCHEMA_BLOCKED
1740
+ assert table.schema_blocking is not True
1741
+ legacy_col = next(c for c in table.columns if c.column == "legacy_flag")
1742
+ assert legacy_col.data_type_status == ValidationStatus.SKIPPED
1743
+ # The column's OTHER checks still ran normally.
1744
+ assert legacy_col.nullable_status in (ValidationStatus.PASS, ValidationStatus.FAIL)
1745
+
1746
+
1747
+ def test_ignore_datatype_columns_does_not_affect_unlisted_columns():
1748
+ """A genuine cross-family type change on a column NOT in
1749
+ ignore_datatype_columns still BLOCKS the table as before - the new
1750
+ field only suppresses the columns explicitly named."""
1751
+ connector = _make_connector()
1752
+ connector.get_table_schema.side_effect = lambda catalog, schema, table: (
1753
+ _schema_df([("id", "int", False), ("name", "string", True)])
1754
+ if catalog == "cat_source"
1755
+ else _schema_df([("id", "int", False), ("name", "int", True)])
1756
+ )
1757
+ validator = CatalogValidator(connector)
1758
+
1759
+ result = validator.compare_catalogs(
1760
+ _request(ignore_datatype_columns=["some_other_column"])
1761
+ )
1762
+
1763
+ table = result.schemas[0].tables[0]
1764
+ assert table.schema_blocking is True
1765
+ assert table.tier_reached == ValidationTier.SCHEMA_BLOCKED
@@ -1107,6 +1107,51 @@ def test_validate_no_primary_key_configured_uses_row_number_fallback(tmp_path: P
1107
1107
  mock_connector.get_row_hashes.assert_not_called()
1108
1108
 
1109
1109
 
1110
+ def test_validate_databricks_wires_only_columns_ignore_columns_and_ignore_datatype(
1111
+ tmp_path: Path,
1112
+ ) -> None:
1113
+ """config.only_columns/ignore_columns/ignore_datatype_columns must
1114
+ reach CatalogValidationRequest unchanged - these were previously
1115
+ silently dropped (the fields didn't exist on ValidatorConfig at all,
1116
+ and _run_databricks_validation never passed them even once added)."""
1117
+ config_path = _full_config(tmp_path)
1118
+ config = load_config(config_path)
1119
+ config.only_columns = ["id", "name"]
1120
+ config.ignore_columns = ["updated_at"]
1121
+ config.ignore_datatype_columns = ["legacy_flag"]
1122
+ save_config(config, config_path)
1123
+ output_path = tmp_path / "validation_report.xlsx"
1124
+
1125
+ mock_connector = _mock_databricks_connector()
1126
+
1127
+ captured_requests = []
1128
+ from table_validator.validators.catalog_validator import CatalogValidator
1129
+
1130
+ real_compare_catalogs = CatalogValidator.compare_catalogs
1131
+
1132
+ def spy_compare_catalogs(self, request):
1133
+ captured_requests.append(request)
1134
+ return real_compare_catalogs(self, request)
1135
+
1136
+ with patch(
1137
+ "table_validator.cli.main.DatabricksConnector", return_value=mock_connector
1138
+ ), patch(
1139
+ "table_validator.cli.main.get_databricks_token", return_value="dapi_fake"
1140
+ ), patch(
1141
+ "table_validator.cli.main.get_azure_credential", return_value=None
1142
+ ), patch.object(CatalogValidator, "compare_catalogs", spy_compare_catalogs):
1143
+ runner.invoke(
1144
+ app,
1145
+ ["validate", "--config-path", str(config_path), "--output", str(output_path)],
1146
+ )
1147
+
1148
+ assert len(captured_requests) == 1
1149
+ request = captured_requests[0]
1150
+ assert request.only_columns == ["id", "name"]
1151
+ assert request.ignore_columns == ["updated_at"]
1152
+ assert request.ignore_datatype_columns == ["legacy_flag"]
1153
+
1154
+
1110
1155
  def test_validate_mode_stats_stops_before_fingerprint_and_row_hash(tmp_path: Path) -> None:
1111
1156
  """--mode stats is the CLI-flag equivalent of the tiered funnel's
1112
1157
  "statistical only" prompt: it must stop after Tier 1 even though the
@@ -1,8 +0,0 @@
1
- {
2
- "tag": "0.1.6",
3
- "distance": 0,
4
- "node": "g9357213ed68fd98fc2c232aed2bb737f273cabbf",
5
- "dirty": false,
6
- "branch": "HEAD",
7
- "node_date": "2026-08-28"
8
- }
File without changes