table-validator 0.1.7__tar.gz → 0.1.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {table_validator-0.1.7/table_validator.egg-info → table_validator-0.1.9}/PKG-INFO +1 -1
  2. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/auth/databricks_auth.py +15 -0
  3. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/main.py +3 -11
  4. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/wizard.py +143 -0
  5. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/config/schema.py +17 -1
  6. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/connectors/databricks_connector.py +86 -30
  7. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/models.py +35 -0
  8. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/catalog_validator.py +400 -105
  9. {table_validator-0.1.7 → table_validator-0.1.9/table_validator.egg-info}/PKG-INFO +1 -1
  10. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/scm_version.json +2 -2
  11. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_catalog_validator.py +599 -6
  12. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_cli.py +7 -4
  13. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_databricks_connector.py +98 -0
  14. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_wizard.py +307 -2
  15. {table_validator-0.1.7 → table_validator-0.1.9}/LICENSE +0 -0
  16. {table_validator-0.1.7 → table_validator-0.1.9}/README.md +0 -0
  17. {table_validator-0.1.7 → table_validator-0.1.9}/pyproject.toml +0 -0
  18. {table_validator-0.1.7 → table_validator-0.1.9}/setup.cfg +0 -0
  19. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/__init__.py +0 -0
  20. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/auth/__init__.py +0 -0
  21. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/auth/azure_auth.py +0 -0
  22. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/__init__.py +0 -0
  23. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/partition_prompt.py +0 -0
  24. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/summary_table.py +0 -0
  25. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/config/__init__.py +0 -0
  26. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/config/manager.py +0 -0
  27. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/connectors/__init__.py +0 -0
  28. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/connectors/azure_connector.py +0 -0
  29. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/engine/__init__.py +0 -0
  30. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/engine/comparison_engine.py +0 -0
  31. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/reports/__init__.py +0 -0
  32. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/reports/excel_report.py +0 -0
  33. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/__init__.py +0 -0
  34. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/blob_discovery.py +0 -0
  35. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/row_validator.py +0 -0
  36. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/SOURCES.txt +0 -0
  37. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/dependency_links.txt +0 -0
  38. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/entry_points.txt +0 -0
  39. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/requires.txt +0 -0
  40. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/scm_file_list.json +0 -0
  41. {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/top_level.txt +0 -0
  42. {table_validator-0.1.7 → table_validator-0.1.9}/tests/__init__.py +0 -0
  43. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_blob_discovery.py +0 -0
  44. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_excel_report.py +0 -0
  45. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_partition_prompt.py +0 -0
  46. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_report_command.py +0 -0
  47. {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_row_validator.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.7
3
+ Version: 0.1.9
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -29,3 +29,18 @@ def get_databricks_token(config: ValidatorConfig, env_path: Path = ENV_PATH) ->
29
29
  """
30
30
  values = dotenv_values(env_path) if env_path.exists() else {}
31
31
  return values.get("DATABRICKS_TOKEN") or None
32
+
33
+
34
+ def host_from_workspace_url(workspace_url: Optional[str]) -> Optional[str]:
35
+ """DatabricksConnector wants a bare hostname; the wizard stores a full
36
+ https:// workspace URL, so strip the scheme and any trailing path.
37
+
38
+ Shared by cli/main.py (building the connector for `validate`) and
39
+ cli/wizard.py (building a connector during `configure` for the
40
+ column-mapping live picker) - kept here rather than in either CLI
41
+ module so neither has to import from the other.
42
+ """
43
+ if not workspace_url:
44
+ return None
45
+ host = workspace_url.replace("https://", "").replace("http://", "")
46
+ return host.split("/")[0]
@@ -10,7 +10,7 @@ from typing import Dict, Optional
10
10
  import typer
11
11
 
12
12
  from table_validator.auth.azure_auth import get_azure_credential
13
- from table_validator.auth.databricks_auth import get_databricks_token
13
+ from table_validator.auth.databricks_auth import get_databricks_token, host_from_workspace_url
14
14
  from table_validator.cli.summary_table import (
15
15
  print_summary_table,
16
16
  summary_from_excel,
@@ -256,7 +256,7 @@ def validate(
256
256
 
257
257
  try:
258
258
  databricks = DatabricksConnector(
259
- host=_host_from_workspace_url(config.databricks.workspace_url),
259
+ host=host_from_workspace_url(config.databricks.workspace_url),
260
260
  token=token,
261
261
  http_path=config.databricks.http_path,
262
262
  )
@@ -473,6 +473,7 @@ def _run_databricks_validation(
473
473
  only_columns=config.only_columns,
474
474
  ignore_columns=config.ignore_columns,
475
475
  ignore_datatype_columns=config.ignore_datatype_columns,
476
+ column_map=config.column_map,
476
477
  )
477
478
 
478
479
  partition_prompt = build_partition_prompt(yes=yes)
@@ -638,15 +639,6 @@ def _missing_config_fields(config: ValidatorConfig) -> list:
638
639
  return missing
639
640
 
640
641
 
641
- def _host_from_workspace_url(workspace_url: Optional[str]) -> Optional[str]:
642
- """DatabricksConnector wants a bare hostname; the wizard stores a full
643
- https:// workspace URL, so strip the scheme and any trailing path."""
644
- if not workspace_url:
645
- return None
646
- host = workspace_url.replace("https://", "").replace("http://", "")
647
- return host.split("/")[0]
648
-
649
-
650
642
  def _open_in_default_app(path: Path) -> None:
651
643
  """
652
644
  Launch `path` in whatever application the OS has associated with its
@@ -117,6 +117,133 @@ def _prompt_primary_key(existing: Optional[list]) -> Optional[list]:
117
117
  return [col.strip() for col in answer.split(",") if col.strip()]
118
118
 
119
119
 
120
+ def _prompt_column_list(prompt_text: str, existing: Optional[list]) -> Optional[list]:
121
+ """Shared free-text parser for a comma-separated column list answer -
122
+ used by all three customization sub-options below. Returns None for
123
+ a blank answer (caller decides the actual default: None vs [])."""
124
+ default_str = ", ".join(existing) if existing else ""
125
+ answer = _ask(questionary.text(prompt_text, default=default_str))
126
+ if not answer:
127
+ return None
128
+ return [col.strip() for col in answer.split(",") if col.strip()]
129
+
130
+
131
+ def _prompt_column_customization(config: ValidatorConfig) -> None:
132
+ """Optional column-level customization, asked right after the
133
+ primary key - only meaningful for the single named table
134
+ (primary_key's same scope). Skipped entirely (leaving any existing
135
+ only_columns/ignore_columns/ignore_datatype_columns untouched) unless
136
+ the user opts in, so a user who never touches this gets identical
137
+ behavior to before this feature existed."""
138
+ customize = questionary.confirm(
139
+ "Customize column validation? (skip specific columns, compare "
140
+ "only specific columns, or ignore datatype mismatches for "
141
+ "specific columns)",
142
+ default=False,
143
+ ).ask()
144
+
145
+ if not customize:
146
+ return
147
+
148
+ config.only_columns = _prompt_column_list(
149
+ "Compare ONLY these columns, comma-separated (leave blank to "
150
+ "compare every common column as usual):",
151
+ config.only_columns,
152
+ )
153
+ config.ignore_columns = _prompt_column_list(
154
+ "SKIP these columns entirely, comma-separated (leave blank to "
155
+ "skip none):",
156
+ config.ignore_columns,
157
+ ) or []
158
+ config.ignore_datatype_columns = _prompt_column_list(
159
+ "Ignore DATATYPE mismatches only for these columns, "
160
+ "comma-separated - their other checks (nullable, statistics, "
161
+ "row values) still run (leave blank to skip none):",
162
+ config.ignore_datatype_columns,
163
+ ) or []
164
+
165
+
166
+ def _prompt_column_mapping(config: ValidatorConfig, secrets: Dict[str, str]) -> None:
167
+ """
168
+ Optional column-name mapping for the single named table (same scope
169
+ as primary_key/column customization) - lets the user pair up
170
+ individual columns that were renamed between source and target (e.g.
171
+ source has 'cust_id', target has 'customer_id').
172
+
173
+ Connects to Databricks LIVE (the first time this wizard ever does so
174
+ during `configure`, rather than only at `validate` time) to fetch
175
+ both tables' real column lists, so the picker can be built from
176
+ actual columns rather than blind free-text entry. Any failure along
177
+ the way (missing/bad credentials, network issue, wrong table name,
178
+ insufficient permissions) is caught broadly and degrades to silently
179
+ skipping this step - `configure` must never crash just because this
180
+ optional, nice-to-have step couldn't reach Databricks. secrets may
181
+ not yet contain a freshly-typed token if the user is configuring for
182
+ the first time in this same run, so DATABRICKS_TOKEN is checked
183
+ there first, falling back to whatever's already on disk.
184
+ """
185
+ from table_validator.auth.databricks_auth import (
186
+ ENV_PATH,
187
+ get_databricks_token,
188
+ host_from_workspace_url,
189
+ )
190
+ from table_validator.connectors.databricks_connector import DatabricksConnector
191
+
192
+ try:
193
+ token = secrets.get("DATABRICKS_TOKEN") or get_databricks_token(config, ENV_PATH)
194
+ if not token:
195
+ return
196
+ databricks = DatabricksConnector(
197
+ host=host_from_workspace_url(config.databricks.workspace_url),
198
+ token=token,
199
+ http_path=config.databricks.http_path,
200
+ )
201
+ source_columns_df = databricks.get_table_schema(
202
+ config.source_table.catalog, config.source_table.schema_name, config.source_table.table
203
+ )
204
+ target_columns_df = databricks.get_table_schema(
205
+ config.target_table.catalog, config.target_table.schema_name, config.target_table.table
206
+ )
207
+ except Exception:
208
+ # Any connection/auth/query failure here just means the live
209
+ # picker isn't available this run - column_map can still be set
210
+ # later by re-running configure, or left unset entirely.
211
+ return
212
+
213
+ source_names = [str(c) for c in source_columns_df["column_name"]]
214
+ target_names = [str(c) for c in target_columns_df["column_name"]]
215
+ target_lower = {t.lower() for t in target_names}
216
+
217
+ unmatched_source = [s for s in source_names if s.lower() not in target_lower]
218
+ if not unmatched_source:
219
+ # Every source column already has an identical-name match -
220
+ # nothing to map, so don't ask anything at all.
221
+ return
222
+
223
+ source_lower = {s.lower() for s in source_names}
224
+ remaining_target = [t for t in target_names if t.lower() not in source_lower]
225
+
226
+ typer.echo(
227
+ f"\n{len(unmatched_source)} source column(s) have no identical-name "
228
+ "match in the target table - map any that were renamed (optional):"
229
+ )
230
+
231
+ new_map: Dict[str, str] = dict(config.column_map or {})
232
+ skip_label = "(skip this column)"
233
+ for src_col in unmatched_source:
234
+ if not remaining_target:
235
+ break
236
+ choice = questionary.select(
237
+ f" Source column '{src_col}' has no matching target column - map it to:",
238
+ choices=remaining_target + [skip_label],
239
+ ).ask()
240
+ if choice and choice != skip_label:
241
+ new_map[src_col] = choice
242
+ remaining_target = [t for t in remaining_target if t != choice]
243
+
244
+ config.column_map = new_map
245
+
246
+
120
247
  def _write_env_file(values: Dict[str, str], env_path: Optional[Path] = None) -> None:
121
248
  """
122
249
  Write secrets to env_path (default: ENV_PATH, resolved at call time)
@@ -394,8 +521,24 @@ def run_configure_wizard() -> None:
394
521
 
395
522
  if source_table_named and config.target_table.table:
396
523
  config.primary_key = _prompt_primary_key(config.primary_key)
524
+ # Column-level customization - same single-table scope as the
525
+ # primary key above (only_columns/ignore_columns/
526
+ # ignore_datatype_columns are only meaningful when there's one
527
+ # specific table to compare). Optional: a user who declines gets
528
+ # identical behavior to before this feature existed.
529
+ typer.echo("\n== Customize validation (optional) ==")
530
+ _prompt_column_customization(config)
531
+ # Column-name mapping (renamed columns) - Databricks-to-Databricks
532
+ # only, since column_map/CatalogValidator don't apply to the
533
+ # Azure Blob/SQL source paths.
534
+ if config.source_type == SourceType.DATABRICKS:
535
+ _prompt_column_mapping(config, secrets)
397
536
  else:
398
537
  config.primary_key = None
538
+ config.only_columns = None
539
+ config.ignore_columns = []
540
+ config.ignore_datatype_columns = []
541
+ config.column_map = {}
399
542
 
400
543
  # ------------------------------------------------------------------
401
544
  # 5. Validations to run
@@ -7,7 +7,7 @@ handled separately in Phase 4 and never appear on these models.
7
7
  from __future__ import annotations
8
8
 
9
9
  from enum import Enum
10
- from typing import List, Optional
10
+ from typing import Dict, List, Optional
11
11
 
12
12
  from pydantic import BaseModel, Field
13
13
 
@@ -202,6 +202,22 @@ class ValidatorConfig(BaseModel):
202
202
  ),
203
203
  )
204
204
 
205
+ column_map: Dict[str, str] = Field(
206
+ default_factory=dict,
207
+ description=(
208
+ "Optional map of source-table column name -> target-table "
209
+ "column name, for an individual column renamed between "
210
+ "source and target (e.g. source has 'cust_id', target has "
211
+ "'customer_id'). Only meaningful for the single named table "
212
+ "in source_table/target_table, same scope as primary_key. "
213
+ "Set interactively by `configure`'s column-mapping picker, "
214
+ "or by hand. A column configured as both a primary key and "
215
+ "an entry here is rejected with a clear error at validate "
216
+ "time - a column used as the row-level join key must have "
217
+ "the identical name on both sides."
218
+ ),
219
+ )
220
+
205
221
  blob_source: BlobSourceConfig = Field(default_factory=BlobSourceConfig)
206
222
  sql_source: SqlSourceConfig = Field(default_factory=SqlSourceConfig)
207
223
 
@@ -878,6 +878,7 @@ class DatabricksConnector:
878
878
  key_values: Sequence[str],
879
879
  value_columns: Sequence[str],
880
880
  limit_samples: int = 500,
881
+ target_value_columns: Optional[Sequence[str]] = None,
881
882
  ) -> List[Dict[str, Any]]:
882
883
  """
883
884
  Tier 5: column-level diff for a bounded, already-known set of
@@ -887,6 +888,13 @@ class DatabricksConnector:
887
888
  column-by-column so callers can report exactly which column(s)
888
889
  differ per row. `key_values` are treated as opaque string literals
889
890
  (matching Tier 4's compare_row_hashes display-key convention).
891
+
892
+ `value_columns` is the SOURCE-side spelling; `target_value_columns`
893
+ (when a column_map applies), the positionally-aligned TARGET-side
894
+ spelling for the same columns. `key_column` itself is assumed
895
+ identical on both sides (a column used as a primary key must not
896
+ also be renamed via column_map - enforced at request-resolution
897
+ time, not here).
890
898
  """
891
899
  if not key_values:
892
900
  return []
@@ -909,6 +917,7 @@ class DatabricksConnector:
909
917
  value_columns=value_columns,
910
918
  changed_query=changed_query,
911
919
  limit_samples=limit_samples,
920
+ target_value_columns=target_value_columns,
912
921
  )
913
922
 
914
923
  def _changed_row_detail(
@@ -921,26 +930,42 @@ class DatabricksConnector:
921
930
  value_columns: Sequence[str],
922
931
  changed_query: str,
923
932
  limit_samples: int,
933
+ target_value_columns: Optional[Sequence[str]] = None,
924
934
  ) -> List[Dict[str, Any]]:
925
935
  """
926
936
  For a bounded sample of changed keys (from `changed_query`), fetch
927
937
  the full source and target rows (key + value columns) plus a
928
938
  whole-row hash for each side, so callers can report exactly which
929
939
  column(s) differ per row without ever collecting a full table.
940
+
941
+ `value_columns` is always the SOURCE-side column spelling.
942
+ `target_value_columns`, when given, is the positionally-aligned
943
+ TARGET-side spelling for the same columns (a column_map case) -
944
+ the source and target SQL each use their own side's names, and
945
+ the result is reconciled back to ONE canonical label per pair
946
+ (the target name, or the shared name when unmapped) so callers
947
+ never have to know which side a given result dict's keys came
948
+ from. When omitted, target_value_columns defaults to
949
+ value_columns (today's behavior, unchanged).
930
950
  """
931
- value_idents = [self._quote_ident(c) for c in value_columns]
932
- all_idents = key_idents + value_idents
933
- select_list = ", ".join(all_idents)
934
- value_concat = ", ".join(value_idents)
951
+ target_value_columns = list(target_value_columns or value_columns)
952
+ source_to_target = dict(zip(value_columns, target_value_columns))
953
+
954
+ source_value_idents = [self._quote_ident(c) for c in value_columns]
955
+ target_value_idents = [self._quote_ident(c) for c in target_value_columns]
956
+ source_select_list = ", ".join(key_idents + source_value_idents)
957
+ target_select_list = ", ".join(key_idents + target_value_idents)
958
+ source_concat = ", ".join(source_value_idents)
959
+ target_concat = ", ".join(target_value_idents)
935
960
 
936
961
  source_rows = self._execute_to_dataframe(f"""
937
- SELECT {select_list}, hash({value_concat}) AS __row_hash
962
+ SELECT {source_select_list}, hash({source_concat}) AS __row_hash
938
963
  FROM {src}
939
964
  WHERE ({key_list}) IN (SELECT {key_list} FROM ({changed_query} LIMIT {int(limit_samples)}) __k)
940
965
  """).to_dict(orient="records")
941
966
 
942
967
  target_rows = self._execute_to_dataframe(f"""
943
- SELECT {select_list}, hash({value_concat}) AS __row_hash
968
+ SELECT {target_select_list}, hash({target_concat}) AS __row_hash
944
969
  FROM {tgt}
945
970
  WHERE ({key_list}) IN (SELECT {key_list} FROM ({changed_query} LIMIT {int(limit_samples)}) __k)
946
971
  """).to_dict(orient="records")
@@ -956,9 +981,12 @@ class DatabricksConnector:
956
981
  if tgt_row is None:
957
982
  continue
958
983
 
984
+ # Diff by pair, but report every result keyed by the
985
+ # canonical (target) column name - src_row/tgt_row are read
986
+ # using each side's own real column name.
959
987
  mismatched_columns = [
960
- col for col in value_columns
961
- if values_differ(src_row.get(col), tgt_row.get(col))
988
+ source_to_target[src_col] for src_col in value_columns
989
+ if values_differ(src_row.get(src_col), tgt_row.get(source_to_target[src_col]))
962
990
  ]
963
991
  if not mismatched_columns:
964
992
  # SQL-side hash() flagged this row as changed, but our
@@ -967,14 +995,21 @@ class DatabricksConnector:
967
995
  # that our value comparison normalizes away). Report the
968
996
  # row anyway rather than silently dropping a row the
969
997
  # mismatch count already accounts for.
970
- mismatched_columns = list(value_columns)
998
+ mismatched_columns = list(target_value_columns)
971
999
 
972
1000
  detail.append(
973
1001
  {
974
1002
  "key": {k: src_row.get(k) for k in key_columns},
975
1003
  "mismatched_columns": mismatched_columns,
976
- "source_values": {c: src_row.get(c) for c in mismatched_columns},
977
- "target_values": {c: tgt_row.get(c) for c in mismatched_columns},
1004
+ "source_values": {
1005
+ source_to_target[src_col]: src_row.get(src_col)
1006
+ for src_col in value_columns
1007
+ if source_to_target[src_col] in mismatched_columns
1008
+ },
1009
+ "target_values": {
1010
+ tgt_col: tgt_row.get(tgt_col)
1011
+ for tgt_col in mismatched_columns
1012
+ },
978
1013
  "source_row_hash": src_row.get("__row_hash"),
979
1014
  "target_row_hash": tgt_row.get("__row_hash"),
980
1015
  }
@@ -1131,6 +1166,8 @@ class DatabricksConnector:
1131
1166
  value_columns: Sequence[str],
1132
1167
  limit_samples: int = 500,
1133
1168
  bucket_predicate: Optional[Tuple[str, Any]] = None,
1169
+ target_order_by_columns: Optional[Sequence[str]] = None,
1170
+ target_value_columns: Optional[Sequence[str]] = None,
1134
1171
  ) -> List[Dict[str, Any]]:
1135
1172
  """
1136
1173
  Best-effort Tier 5 column-level diff for the ROW_NUMBER() fallback
@@ -1139,12 +1176,15 @@ class DatabricksConnector:
1139
1176
  used by get_row_hashes_by_row_number, filtered down to the given
1140
1177
  row numbers, then diffs the fetched rows column-by-column.
1141
1178
 
1142
- `order_by_columns` MUST be the exact same column list (same order)
1143
- passed to get_row_hashes_by_row_number for this table - that is
1144
- what keeps row numbers consistent between the hash-computation
1145
- pass (Tier 4) and this re-fetch (Tier 5). `value_columns` is
1146
- normally the same list too, since the row-number fallback has no
1147
- key to exclude.
1179
+ `order_by_columns`/`value_columns` are the SOURCE-side spelling;
1180
+ `target_order_by_columns`/`target_value_columns` (when a
1181
+ column_map applies), the positionally-aligned TARGET-side
1182
+ spelling for the same columns - each MUST be the exact same
1183
+ per-side column list (same order) passed to
1184
+ get_row_hashes_by_row_number for this table, which is what keeps
1185
+ row numbers consistent between the hash-computation pass (Tier 4)
1186
+ and this re-fetch (Tier 5). When omitted, the target lists default
1187
+ to the source lists (today's behavior, unchanged).
1148
1188
 
1149
1189
  This is inherently best-effort, not a substitute for a real key:
1150
1190
  "row N" on the source and target are only the same logical record
@@ -1161,25 +1201,34 @@ class DatabricksConnector:
1161
1201
  Returns the same shape as _changed_row_detail: one dict per
1162
1202
  row with "key" (here always {"row_number": N}),
1163
1203
  "mismatched_columns", "source_values", "target_values",
1164
- "source_row_hash", "target_row_hash".
1204
+ "source_row_hash", "target_row_hash" - all keyed/labeled by the
1205
+ canonical (target) column name.
1165
1206
  """
1166
1207
  if not row_numbers:
1167
1208
  return []
1168
1209
 
1210
+ target_order_by_columns = list(target_order_by_columns or order_by_columns)
1211
+ target_value_columns = list(target_value_columns or value_columns)
1212
+ source_to_target = dict(zip(value_columns, target_value_columns))
1213
+
1169
1214
  src = self._qualify(source_catalog, schema, table)
1170
1215
  tgt = self._qualify(target_catalog, schema, table)
1171
- order_by = ", ".join(self._quote_ident(c) for c in order_by_columns)
1172
- value_idents = [self._quote_ident(c) for c in value_columns]
1173
- select_list = ", ".join(value_idents)
1216
+ source_order_by = ", ".join(self._quote_ident(c) for c in order_by_columns)
1217
+ target_order_by = ", ".join(self._quote_ident(c) for c in target_order_by_columns)
1218
+ source_value_idents = [self._quote_ident(c) for c in value_columns]
1219
+ target_value_idents = [self._quote_ident(c) for c in target_value_columns]
1220
+ source_select_list = ", ".join(source_value_idents)
1221
+ target_select_list = ", ".join(target_value_idents)
1174
1222
  where_clause = self._bucket_where_clause(bucket_predicate)
1175
- row_hash_expr = self._row_hash_expr(value_columns)
1223
+ source_row_hash_expr = self._row_hash_expr(value_columns)
1224
+ target_row_hash_expr = self._row_hash_expr(target_value_columns)
1176
1225
 
1177
1226
  # row_numbers are Python ints derived from our own prior
1178
1227
  # ROW_NUMBER() output (never user input) - safe to inline.
1179
1228
  unique_row_numbers = sorted(set(int(n) for n in row_numbers))[: int(limit_samples)]
1180
1229
  row_numbers_csv = ", ".join(str(n) for n in unique_row_numbers)
1181
1230
 
1182
- def _numbered_query(fqtn: str) -> str:
1231
+ def _numbered_query(fqtn: str, order_by: str, select_list: str, row_hash_expr: str) -> str:
1183
1232
  return f"""
1184
1233
  SELECT row_number, {select_list}, {row_hash_expr} AS __row_hash
1185
1234
  FROM (
@@ -1194,10 +1243,10 @@ class DatabricksConnector:
1194
1243
 
1195
1244
  try:
1196
1245
  source_rows = self._execute_to_dataframe(
1197
- _numbered_query(src)
1246
+ _numbered_query(src, source_order_by, source_select_list, source_row_hash_expr)
1198
1247
  ).to_dict(orient="records")
1199
1248
  target_rows = self._execute_to_dataframe(
1200
- _numbered_query(tgt)
1249
+ _numbered_query(tgt, target_order_by, target_select_list, target_row_hash_expr)
1201
1250
  ).to_dict(orient="records")
1202
1251
  except Exception as exc:
1203
1252
  logger.exception(
@@ -1216,18 +1265,25 @@ class DatabricksConnector:
1216
1265
  continue
1217
1266
 
1218
1267
  mismatched_columns = [
1219
- col for col in value_columns
1220
- if values_differ(src_row.get(col), tgt_row.get(col))
1268
+ source_to_target[src_col] for src_col in value_columns
1269
+ if values_differ(src_row.get(src_col), tgt_row.get(source_to_target[src_col]))
1221
1270
  ]
1222
1271
  if not mismatched_columns:
1223
- mismatched_columns = list(value_columns)
1272
+ mismatched_columns = list(target_value_columns)
1224
1273
 
1225
1274
  detail.append(
1226
1275
  {
1227
1276
  "key": {"row_number": src_row["row_number"]},
1228
1277
  "mismatched_columns": mismatched_columns,
1229
- "source_values": {c: src_row.get(c) for c in mismatched_columns},
1230
- "target_values": {c: tgt_row.get(c) for c in mismatched_columns},
1278
+ "source_values": {
1279
+ source_to_target[src_col]: src_row.get(src_col)
1280
+ for src_col in value_columns
1281
+ if source_to_target[src_col] in mismatched_columns
1282
+ },
1283
+ "target_values": {
1284
+ tgt_col: tgt_row.get(tgt_col)
1285
+ for tgt_col in mismatched_columns
1286
+ },
1231
1287
  "source_row_hash": src_row.get("__row_hash"),
1232
1288
  "target_row_hash": tgt_row.get("__row_hash"),
1233
1289
  }
@@ -517,6 +517,30 @@ class CatalogValidationRequest(BaseModel):
517
517
  ),
518
518
  )
519
519
 
520
+ column_map: Dict[str, str] = Field(
521
+ default_factory=dict,
522
+ description=(
523
+ "Optional map of source-table column name -> target-table "
524
+ "column name, for when an individual column was renamed "
525
+ "between source and target (e.g. source has 'cust_id', "
526
+ "target has 'customer_id'). An explicit pair like this is "
527
+ "treated as fully equivalent through the entire pipeline - "
528
+ "schema/type/nullable checks, null/distinct/min-max "
529
+ "statistics, whole-table fingerprint, row-hash diff, and "
530
+ "column-level mismatch detail - bypassing name-based column "
531
+ "matching entirely for that pair (unlike ignore_columns/"
532
+ "only_columns, which still require the name to appear in the "
533
+ "intersection). Unmapped columns are matched by identical "
534
+ "name as usual. Resolved before only_columns/ignore_columns/"
535
+ "ignore_datatype_columns apply, so those three act on the "
536
+ "resolved/canonical (target-side) name. A mapped name that "
537
+ "doesn't actually exist on either side produces a clear "
538
+ "error rather than a silent no-op. Known limitation: a "
539
+ "column configured as a primary key, or used as a Tier 3 "
540
+ "partition/bucket column, must not also appear here."
541
+ ),
542
+ )
543
+
520
544
  case_sensitive_columns: bool = Field(
521
545
  default=False,
522
546
  description="Case-sensitive column name comparison.",
@@ -758,6 +782,12 @@ class ColumnValidationResult(BaseModel):
758
782
  column: str
759
783
  status: ValidationStatus
760
784
 
785
+ # Populated only when column_map actually renamed this column (source
786
+ # name differs from the target/canonical name shown above) - mirrors
787
+ # TableValidationResult.source_table_name's "only set when it
788
+ # differs" convention.
789
+ source_column: Optional[str] = None
790
+
761
791
  source_data_type: Optional[str] = None
762
792
  target_data_type: Optional[str] = None
763
793
  data_type_status: Optional[ValidationStatus] = None
@@ -799,6 +829,11 @@ class RowMismatchDetail(BaseModel):
799
829
  primary_key: Dict[str, Any] = Field(default_factory=dict)
800
830
  mismatch_column: str
801
831
 
832
+ # Populated only when column_map actually renamed this column -
833
+ # mismatch_column stays the canonical/target-side name (existing
834
+ # report rendering, existing tests, unaffected).
835
+ source_mismatch_column: Optional[str] = None
836
+
802
837
  source_value: Optional[Any] = None
803
838
  target_value: Optional[Any] = None
804
839