table-validator 0.1.7__tar.gz → 0.1.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {table_validator-0.1.7/table_validator.egg-info → table_validator-0.1.9}/PKG-INFO +1 -1
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/auth/databricks_auth.py +15 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/main.py +3 -11
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/wizard.py +143 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/config/schema.py +17 -1
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/connectors/databricks_connector.py +86 -30
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/models.py +35 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/catalog_validator.py +400 -105
- {table_validator-0.1.7 → table_validator-0.1.9/table_validator.egg-info}/PKG-INFO +1 -1
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/scm_version.json +2 -2
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_catalog_validator.py +599 -6
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_cli.py +7 -4
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_databricks_connector.py +98 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_wizard.py +307 -2
- {table_validator-0.1.7 → table_validator-0.1.9}/LICENSE +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/README.md +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/pyproject.toml +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/setup.cfg +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/auth/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/auth/azure_auth.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/partition_prompt.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/cli/summary_table.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/config/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/config/manager.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/connectors/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/connectors/azure_connector.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/engine/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/engine/comparison_engine.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/reports/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/reports/excel_report.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/blob_discovery.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator/validators/row_validator.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/SOURCES.txt +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/dependency_links.txt +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/entry_points.txt +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/requires.txt +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/scm_file_list.json +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/table_validator.egg-info/top_level.txt +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/__init__.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_blob_discovery.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_excel_report.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_partition_prompt.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_report_command.py +0 -0
- {table_validator-0.1.7 → table_validator-0.1.9}/tests/test_row_validator.py +0 -0
|
@@ -29,3 +29,18 @@ def get_databricks_token(config: ValidatorConfig, env_path: Path = ENV_PATH) ->
|
|
|
29
29
|
"""
|
|
30
30
|
values = dotenv_values(env_path) if env_path.exists() else {}
|
|
31
31
|
return values.get("DATABRICKS_TOKEN") or None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def host_from_workspace_url(workspace_url: Optional[str]) -> Optional[str]:
|
|
35
|
+
"""DatabricksConnector wants a bare hostname; the wizard stores a full
|
|
36
|
+
https:// workspace URL, so strip the scheme and any trailing path.
|
|
37
|
+
|
|
38
|
+
Shared by cli/main.py (building the connector for `validate`) and
|
|
39
|
+
cli/wizard.py (building a connector during `configure` for the
|
|
40
|
+
column-mapping live picker) - kept here rather than in either CLI
|
|
41
|
+
module so neither has to import from the other.
|
|
42
|
+
"""
|
|
43
|
+
if not workspace_url:
|
|
44
|
+
return None
|
|
45
|
+
host = workspace_url.replace("https://", "").replace("http://", "")
|
|
46
|
+
return host.split("/")[0]
|
|
@@ -10,7 +10,7 @@ from typing import Dict, Optional
|
|
|
10
10
|
import typer
|
|
11
11
|
|
|
12
12
|
from table_validator.auth.azure_auth import get_azure_credential
|
|
13
|
-
from table_validator.auth.databricks_auth import get_databricks_token
|
|
13
|
+
from table_validator.auth.databricks_auth import get_databricks_token, host_from_workspace_url
|
|
14
14
|
from table_validator.cli.summary_table import (
|
|
15
15
|
print_summary_table,
|
|
16
16
|
summary_from_excel,
|
|
@@ -256,7 +256,7 @@ def validate(
|
|
|
256
256
|
|
|
257
257
|
try:
|
|
258
258
|
databricks = DatabricksConnector(
|
|
259
|
-
host=
|
|
259
|
+
host=host_from_workspace_url(config.databricks.workspace_url),
|
|
260
260
|
token=token,
|
|
261
261
|
http_path=config.databricks.http_path,
|
|
262
262
|
)
|
|
@@ -473,6 +473,7 @@ def _run_databricks_validation(
|
|
|
473
473
|
only_columns=config.only_columns,
|
|
474
474
|
ignore_columns=config.ignore_columns,
|
|
475
475
|
ignore_datatype_columns=config.ignore_datatype_columns,
|
|
476
|
+
column_map=config.column_map,
|
|
476
477
|
)
|
|
477
478
|
|
|
478
479
|
partition_prompt = build_partition_prompt(yes=yes)
|
|
@@ -638,15 +639,6 @@ def _missing_config_fields(config: ValidatorConfig) -> list:
|
|
|
638
639
|
return missing
|
|
639
640
|
|
|
640
641
|
|
|
641
|
-
def _host_from_workspace_url(workspace_url: Optional[str]) -> Optional[str]:
|
|
642
|
-
"""DatabricksConnector wants a bare hostname; the wizard stores a full
|
|
643
|
-
https:// workspace URL, so strip the scheme and any trailing path."""
|
|
644
|
-
if not workspace_url:
|
|
645
|
-
return None
|
|
646
|
-
host = workspace_url.replace("https://", "").replace("http://", "")
|
|
647
|
-
return host.split("/")[0]
|
|
648
|
-
|
|
649
|
-
|
|
650
642
|
def _open_in_default_app(path: Path) -> None:
|
|
651
643
|
"""
|
|
652
644
|
Launch `path` in whatever application the OS has associated with its
|
|
@@ -117,6 +117,133 @@ def _prompt_primary_key(existing: Optional[list]) -> Optional[list]:
|
|
|
117
117
|
return [col.strip() for col in answer.split(",") if col.strip()]
|
|
118
118
|
|
|
119
119
|
|
|
120
|
+
def _prompt_column_list(prompt_text: str, existing: Optional[list]) -> Optional[list]:
|
|
121
|
+
"""Shared free-text parser for a comma-separated column list answer -
|
|
122
|
+
used by all three customization sub-options below. Returns None for
|
|
123
|
+
a blank answer (caller decides the actual default: None vs [])."""
|
|
124
|
+
default_str = ", ".join(existing) if existing else ""
|
|
125
|
+
answer = _ask(questionary.text(prompt_text, default=default_str))
|
|
126
|
+
if not answer:
|
|
127
|
+
return None
|
|
128
|
+
return [col.strip() for col in answer.split(",") if col.strip()]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _prompt_column_customization(config: ValidatorConfig) -> None:
|
|
132
|
+
"""Optional column-level customization, asked right after the
|
|
133
|
+
primary key - only meaningful for the single named table
|
|
134
|
+
(primary_key's same scope). Skipped entirely (leaving any existing
|
|
135
|
+
only_columns/ignore_columns/ignore_datatype_columns untouched) unless
|
|
136
|
+
the user opts in, so a user who never touches this gets identical
|
|
137
|
+
behavior to before this feature existed."""
|
|
138
|
+
customize = questionary.confirm(
|
|
139
|
+
"Customize column validation? (skip specific columns, compare "
|
|
140
|
+
"only specific columns, or ignore datatype mismatches for "
|
|
141
|
+
"specific columns)",
|
|
142
|
+
default=False,
|
|
143
|
+
).ask()
|
|
144
|
+
|
|
145
|
+
if not customize:
|
|
146
|
+
return
|
|
147
|
+
|
|
148
|
+
config.only_columns = _prompt_column_list(
|
|
149
|
+
"Compare ONLY these columns, comma-separated (leave blank to "
|
|
150
|
+
"compare every common column as usual):",
|
|
151
|
+
config.only_columns,
|
|
152
|
+
)
|
|
153
|
+
config.ignore_columns = _prompt_column_list(
|
|
154
|
+
"SKIP these columns entirely, comma-separated (leave blank to "
|
|
155
|
+
"skip none):",
|
|
156
|
+
config.ignore_columns,
|
|
157
|
+
) or []
|
|
158
|
+
config.ignore_datatype_columns = _prompt_column_list(
|
|
159
|
+
"Ignore DATATYPE mismatches only for these columns, "
|
|
160
|
+
"comma-separated - their other checks (nullable, statistics, "
|
|
161
|
+
"row values) still run (leave blank to skip none):",
|
|
162
|
+
config.ignore_datatype_columns,
|
|
163
|
+
) or []
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _prompt_column_mapping(config: ValidatorConfig, secrets: Dict[str, str]) -> None:
|
|
167
|
+
"""
|
|
168
|
+
Optional column-name mapping for the single named table (same scope
|
|
169
|
+
as primary_key/column customization) - lets the user pair up
|
|
170
|
+
individual columns that were renamed between source and target (e.g.
|
|
171
|
+
source has 'cust_id', target has 'customer_id').
|
|
172
|
+
|
|
173
|
+
Connects to Databricks LIVE (the first time this wizard ever does so
|
|
174
|
+
during `configure`, rather than only at `validate` time) to fetch
|
|
175
|
+
both tables' real column lists, so the picker can be built from
|
|
176
|
+
actual columns rather than blind free-text entry. Any failure along
|
|
177
|
+
the way (missing/bad credentials, network issue, wrong table name,
|
|
178
|
+
insufficient permissions) is caught broadly and degrades to silently
|
|
179
|
+
skipping this step - `configure` must never crash just because this
|
|
180
|
+
optional, nice-to-have step couldn't reach Databricks. secrets may
|
|
181
|
+
not yet contain a freshly-typed token if the user is configuring for
|
|
182
|
+
the first time in this same run, so DATABRICKS_TOKEN is checked
|
|
183
|
+
there first, falling back to whatever's already on disk.
|
|
184
|
+
"""
|
|
185
|
+
from table_validator.auth.databricks_auth import (
|
|
186
|
+
ENV_PATH,
|
|
187
|
+
get_databricks_token,
|
|
188
|
+
host_from_workspace_url,
|
|
189
|
+
)
|
|
190
|
+
from table_validator.connectors.databricks_connector import DatabricksConnector
|
|
191
|
+
|
|
192
|
+
try:
|
|
193
|
+
token = secrets.get("DATABRICKS_TOKEN") or get_databricks_token(config, ENV_PATH)
|
|
194
|
+
if not token:
|
|
195
|
+
return
|
|
196
|
+
databricks = DatabricksConnector(
|
|
197
|
+
host=host_from_workspace_url(config.databricks.workspace_url),
|
|
198
|
+
token=token,
|
|
199
|
+
http_path=config.databricks.http_path,
|
|
200
|
+
)
|
|
201
|
+
source_columns_df = databricks.get_table_schema(
|
|
202
|
+
config.source_table.catalog, config.source_table.schema_name, config.source_table.table
|
|
203
|
+
)
|
|
204
|
+
target_columns_df = databricks.get_table_schema(
|
|
205
|
+
config.target_table.catalog, config.target_table.schema_name, config.target_table.table
|
|
206
|
+
)
|
|
207
|
+
except Exception:
|
|
208
|
+
# Any connection/auth/query failure here just means the live
|
|
209
|
+
# picker isn't available this run - column_map can still be set
|
|
210
|
+
# later by re-running configure, or left unset entirely.
|
|
211
|
+
return
|
|
212
|
+
|
|
213
|
+
source_names = [str(c) for c in source_columns_df["column_name"]]
|
|
214
|
+
target_names = [str(c) for c in target_columns_df["column_name"]]
|
|
215
|
+
target_lower = {t.lower() for t in target_names}
|
|
216
|
+
|
|
217
|
+
unmatched_source = [s for s in source_names if s.lower() not in target_lower]
|
|
218
|
+
if not unmatched_source:
|
|
219
|
+
# Every source column already has an identical-name match -
|
|
220
|
+
# nothing to map, so don't ask anything at all.
|
|
221
|
+
return
|
|
222
|
+
|
|
223
|
+
source_lower = {s.lower() for s in source_names}
|
|
224
|
+
remaining_target = [t for t in target_names if t.lower() not in source_lower]
|
|
225
|
+
|
|
226
|
+
typer.echo(
|
|
227
|
+
f"\n{len(unmatched_source)} source column(s) have no identical-name "
|
|
228
|
+
"match in the target table - map any that were renamed (optional):"
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
new_map: Dict[str, str] = dict(config.column_map or {})
|
|
232
|
+
skip_label = "(skip this column)"
|
|
233
|
+
for src_col in unmatched_source:
|
|
234
|
+
if not remaining_target:
|
|
235
|
+
break
|
|
236
|
+
choice = questionary.select(
|
|
237
|
+
f" Source column '{src_col}' has no matching target column - map it to:",
|
|
238
|
+
choices=remaining_target + [skip_label],
|
|
239
|
+
).ask()
|
|
240
|
+
if choice and choice != skip_label:
|
|
241
|
+
new_map[src_col] = choice
|
|
242
|
+
remaining_target = [t for t in remaining_target if t != choice]
|
|
243
|
+
|
|
244
|
+
config.column_map = new_map
|
|
245
|
+
|
|
246
|
+
|
|
120
247
|
def _write_env_file(values: Dict[str, str], env_path: Optional[Path] = None) -> None:
|
|
121
248
|
"""
|
|
122
249
|
Write secrets to env_path (default: ENV_PATH, resolved at call time)
|
|
@@ -394,8 +521,24 @@ def run_configure_wizard() -> None:
|
|
|
394
521
|
|
|
395
522
|
if source_table_named and config.target_table.table:
|
|
396
523
|
config.primary_key = _prompt_primary_key(config.primary_key)
|
|
524
|
+
# Column-level customization - same single-table scope as the
|
|
525
|
+
# primary key above (only_columns/ignore_columns/
|
|
526
|
+
# ignore_datatype_columns are only meaningful when there's one
|
|
527
|
+
# specific table to compare). Optional: a user who declines gets
|
|
528
|
+
# identical behavior to before this feature existed.
|
|
529
|
+
typer.echo("\n== Customize validation (optional) ==")
|
|
530
|
+
_prompt_column_customization(config)
|
|
531
|
+
# Column-name mapping (renamed columns) - Databricks-to-Databricks
|
|
532
|
+
# only, since column_map/CatalogValidator don't apply to the
|
|
533
|
+
# Azure Blob/SQL source paths.
|
|
534
|
+
if config.source_type == SourceType.DATABRICKS:
|
|
535
|
+
_prompt_column_mapping(config, secrets)
|
|
397
536
|
else:
|
|
398
537
|
config.primary_key = None
|
|
538
|
+
config.only_columns = None
|
|
539
|
+
config.ignore_columns = []
|
|
540
|
+
config.ignore_datatype_columns = []
|
|
541
|
+
config.column_map = {}
|
|
399
542
|
|
|
400
543
|
# ------------------------------------------------------------------
|
|
401
544
|
# 5. Validations to run
|
|
@@ -7,7 +7,7 @@ handled separately in Phase 4 and never appear on these models.
|
|
|
7
7
|
from __future__ import annotations
|
|
8
8
|
|
|
9
9
|
from enum import Enum
|
|
10
|
-
from typing import List, Optional
|
|
10
|
+
from typing import Dict, List, Optional
|
|
11
11
|
|
|
12
12
|
from pydantic import BaseModel, Field
|
|
13
13
|
|
|
@@ -202,6 +202,22 @@ class ValidatorConfig(BaseModel):
|
|
|
202
202
|
),
|
|
203
203
|
)
|
|
204
204
|
|
|
205
|
+
column_map: Dict[str, str] = Field(
|
|
206
|
+
default_factory=dict,
|
|
207
|
+
description=(
|
|
208
|
+
"Optional map of source-table column name -> target-table "
|
|
209
|
+
"column name, for an individual column renamed between "
|
|
210
|
+
"source and target (e.g. source has 'cust_id', target has "
|
|
211
|
+
"'customer_id'). Only meaningful for the single named table "
|
|
212
|
+
"in source_table/target_table, same scope as primary_key. "
|
|
213
|
+
"Set interactively by `configure`'s column-mapping picker, "
|
|
214
|
+
"or by hand. A column configured as both a primary key and "
|
|
215
|
+
"an entry here is rejected with a clear error at validate "
|
|
216
|
+
"time - a column used as the row-level join key must have "
|
|
217
|
+
"the identical name on both sides."
|
|
218
|
+
),
|
|
219
|
+
)
|
|
220
|
+
|
|
205
221
|
blob_source: BlobSourceConfig = Field(default_factory=BlobSourceConfig)
|
|
206
222
|
sql_source: SqlSourceConfig = Field(default_factory=SqlSourceConfig)
|
|
207
223
|
|
{table_validator-0.1.7 → table_validator-0.1.9}/table_validator/connectors/databricks_connector.py
RENAMED
|
@@ -878,6 +878,7 @@ class DatabricksConnector:
|
|
|
878
878
|
key_values: Sequence[str],
|
|
879
879
|
value_columns: Sequence[str],
|
|
880
880
|
limit_samples: int = 500,
|
|
881
|
+
target_value_columns: Optional[Sequence[str]] = None,
|
|
881
882
|
) -> List[Dict[str, Any]]:
|
|
882
883
|
"""
|
|
883
884
|
Tier 5: column-level diff for a bounded, already-known set of
|
|
@@ -887,6 +888,13 @@ class DatabricksConnector:
|
|
|
887
888
|
column-by-column so callers can report exactly which column(s)
|
|
888
889
|
differ per row. `key_values` are treated as opaque string literals
|
|
889
890
|
(matching Tier 4's compare_row_hashes display-key convention).
|
|
891
|
+
|
|
892
|
+
`value_columns` is the SOURCE-side spelling; `target_value_columns`
|
|
893
|
+
(when a column_map applies), the positionally-aligned TARGET-side
|
|
894
|
+
spelling for the same columns. `key_column` itself is assumed
|
|
895
|
+
identical on both sides (a column used as a primary key must not
|
|
896
|
+
also be renamed via column_map - enforced at request-resolution
|
|
897
|
+
time, not here).
|
|
890
898
|
"""
|
|
891
899
|
if not key_values:
|
|
892
900
|
return []
|
|
@@ -909,6 +917,7 @@ class DatabricksConnector:
|
|
|
909
917
|
value_columns=value_columns,
|
|
910
918
|
changed_query=changed_query,
|
|
911
919
|
limit_samples=limit_samples,
|
|
920
|
+
target_value_columns=target_value_columns,
|
|
912
921
|
)
|
|
913
922
|
|
|
914
923
|
def _changed_row_detail(
|
|
@@ -921,26 +930,42 @@ class DatabricksConnector:
|
|
|
921
930
|
value_columns: Sequence[str],
|
|
922
931
|
changed_query: str,
|
|
923
932
|
limit_samples: int,
|
|
933
|
+
target_value_columns: Optional[Sequence[str]] = None,
|
|
924
934
|
) -> List[Dict[str, Any]]:
|
|
925
935
|
"""
|
|
926
936
|
For a bounded sample of changed keys (from `changed_query`), fetch
|
|
927
937
|
the full source and target rows (key + value columns) plus a
|
|
928
938
|
whole-row hash for each side, so callers can report exactly which
|
|
929
939
|
column(s) differ per row without ever collecting a full table.
|
|
940
|
+
|
|
941
|
+
`value_columns` is always the SOURCE-side column spelling.
|
|
942
|
+
`target_value_columns`, when given, is the positionally-aligned
|
|
943
|
+
TARGET-side spelling for the same columns (a column_map case) -
|
|
944
|
+
the source and target SQL each use their own side's names, and
|
|
945
|
+
the result is reconciled back to ONE canonical label per pair
|
|
946
|
+
(the target name, or the shared name when unmapped) so callers
|
|
947
|
+
never have to know which side a given result dict's keys came
|
|
948
|
+
from. When omitted, target_value_columns defaults to
|
|
949
|
+
value_columns (today's behavior, unchanged).
|
|
930
950
|
"""
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
951
|
+
target_value_columns = list(target_value_columns or value_columns)
|
|
952
|
+
source_to_target = dict(zip(value_columns, target_value_columns))
|
|
953
|
+
|
|
954
|
+
source_value_idents = [self._quote_ident(c) for c in value_columns]
|
|
955
|
+
target_value_idents = [self._quote_ident(c) for c in target_value_columns]
|
|
956
|
+
source_select_list = ", ".join(key_idents + source_value_idents)
|
|
957
|
+
target_select_list = ", ".join(key_idents + target_value_idents)
|
|
958
|
+
source_concat = ", ".join(source_value_idents)
|
|
959
|
+
target_concat = ", ".join(target_value_idents)
|
|
935
960
|
|
|
936
961
|
source_rows = self._execute_to_dataframe(f"""
|
|
937
|
-
SELECT {
|
|
962
|
+
SELECT {source_select_list}, hash({source_concat}) AS __row_hash
|
|
938
963
|
FROM {src}
|
|
939
964
|
WHERE ({key_list}) IN (SELECT {key_list} FROM ({changed_query} LIMIT {int(limit_samples)}) __k)
|
|
940
965
|
""").to_dict(orient="records")
|
|
941
966
|
|
|
942
967
|
target_rows = self._execute_to_dataframe(f"""
|
|
943
|
-
SELECT {
|
|
968
|
+
SELECT {target_select_list}, hash({target_concat}) AS __row_hash
|
|
944
969
|
FROM {tgt}
|
|
945
970
|
WHERE ({key_list}) IN (SELECT {key_list} FROM ({changed_query} LIMIT {int(limit_samples)}) __k)
|
|
946
971
|
""").to_dict(orient="records")
|
|
@@ -956,9 +981,12 @@ class DatabricksConnector:
|
|
|
956
981
|
if tgt_row is None:
|
|
957
982
|
continue
|
|
958
983
|
|
|
984
|
+
# Diff by pair, but report every result keyed by the
|
|
985
|
+
# canonical (target) column name - src_row/tgt_row are read
|
|
986
|
+
# using each side's own real column name.
|
|
959
987
|
mismatched_columns = [
|
|
960
|
-
|
|
961
|
-
if values_differ(src_row.get(
|
|
988
|
+
source_to_target[src_col] for src_col in value_columns
|
|
989
|
+
if values_differ(src_row.get(src_col), tgt_row.get(source_to_target[src_col]))
|
|
962
990
|
]
|
|
963
991
|
if not mismatched_columns:
|
|
964
992
|
# SQL-side hash() flagged this row as changed, but our
|
|
@@ -967,14 +995,21 @@ class DatabricksConnector:
|
|
|
967
995
|
# that our value comparison normalizes away). Report the
|
|
968
996
|
# row anyway rather than silently dropping a row the
|
|
969
997
|
# mismatch count already accounts for.
|
|
970
|
-
mismatched_columns = list(
|
|
998
|
+
mismatched_columns = list(target_value_columns)
|
|
971
999
|
|
|
972
1000
|
detail.append(
|
|
973
1001
|
{
|
|
974
1002
|
"key": {k: src_row.get(k) for k in key_columns},
|
|
975
1003
|
"mismatched_columns": mismatched_columns,
|
|
976
|
-
"source_values": {
|
|
977
|
-
|
|
1004
|
+
"source_values": {
|
|
1005
|
+
source_to_target[src_col]: src_row.get(src_col)
|
|
1006
|
+
for src_col in value_columns
|
|
1007
|
+
if source_to_target[src_col] in mismatched_columns
|
|
1008
|
+
},
|
|
1009
|
+
"target_values": {
|
|
1010
|
+
tgt_col: tgt_row.get(tgt_col)
|
|
1011
|
+
for tgt_col in mismatched_columns
|
|
1012
|
+
},
|
|
978
1013
|
"source_row_hash": src_row.get("__row_hash"),
|
|
979
1014
|
"target_row_hash": tgt_row.get("__row_hash"),
|
|
980
1015
|
}
|
|
@@ -1131,6 +1166,8 @@ class DatabricksConnector:
|
|
|
1131
1166
|
value_columns: Sequence[str],
|
|
1132
1167
|
limit_samples: int = 500,
|
|
1133
1168
|
bucket_predicate: Optional[Tuple[str, Any]] = None,
|
|
1169
|
+
target_order_by_columns: Optional[Sequence[str]] = None,
|
|
1170
|
+
target_value_columns: Optional[Sequence[str]] = None,
|
|
1134
1171
|
) -> List[Dict[str, Any]]:
|
|
1135
1172
|
"""
|
|
1136
1173
|
Best-effort Tier 5 column-level diff for the ROW_NUMBER() fallback
|
|
@@ -1139,12 +1176,15 @@ class DatabricksConnector:
|
|
|
1139
1176
|
used by get_row_hashes_by_row_number, filtered down to the given
|
|
1140
1177
|
row numbers, then diffs the fetched rows column-by-column.
|
|
1141
1178
|
|
|
1142
|
-
`order_by_columns`
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1179
|
+
`order_by_columns`/`value_columns` are the SOURCE-side spelling;
|
|
1180
|
+
`target_order_by_columns`/`target_value_columns` (when a
|
|
1181
|
+
column_map applies), the positionally-aligned TARGET-side
|
|
1182
|
+
spelling for the same columns - each MUST be the exact same
|
|
1183
|
+
per-side column list (same order) passed to
|
|
1184
|
+
get_row_hashes_by_row_number for this table, which is what keeps
|
|
1185
|
+
row numbers consistent between the hash-computation pass (Tier 4)
|
|
1186
|
+
and this re-fetch (Tier 5). When omitted, the target lists default
|
|
1187
|
+
to the source lists (today's behavior, unchanged).
|
|
1148
1188
|
|
|
1149
1189
|
This is inherently best-effort, not a substitute for a real key:
|
|
1150
1190
|
"row N" on the source and target are only the same logical record
|
|
@@ -1161,25 +1201,34 @@ class DatabricksConnector:
|
|
|
1161
1201
|
Returns the same shape as _changed_row_detail: one dict per
|
|
1162
1202
|
row with "key" (here always {"row_number": N}),
|
|
1163
1203
|
"mismatched_columns", "source_values", "target_values",
|
|
1164
|
-
"source_row_hash", "target_row_hash"
|
|
1204
|
+
"source_row_hash", "target_row_hash" - all keyed/labeled by the
|
|
1205
|
+
canonical (target) column name.
|
|
1165
1206
|
"""
|
|
1166
1207
|
if not row_numbers:
|
|
1167
1208
|
return []
|
|
1168
1209
|
|
|
1210
|
+
target_order_by_columns = list(target_order_by_columns or order_by_columns)
|
|
1211
|
+
target_value_columns = list(target_value_columns or value_columns)
|
|
1212
|
+
source_to_target = dict(zip(value_columns, target_value_columns))
|
|
1213
|
+
|
|
1169
1214
|
src = self._qualify(source_catalog, schema, table)
|
|
1170
1215
|
tgt = self._qualify(target_catalog, schema, table)
|
|
1171
|
-
|
|
1172
|
-
|
|
1173
|
-
|
|
1216
|
+
source_order_by = ", ".join(self._quote_ident(c) for c in order_by_columns)
|
|
1217
|
+
target_order_by = ", ".join(self._quote_ident(c) for c in target_order_by_columns)
|
|
1218
|
+
source_value_idents = [self._quote_ident(c) for c in value_columns]
|
|
1219
|
+
target_value_idents = [self._quote_ident(c) for c in target_value_columns]
|
|
1220
|
+
source_select_list = ", ".join(source_value_idents)
|
|
1221
|
+
target_select_list = ", ".join(target_value_idents)
|
|
1174
1222
|
where_clause = self._bucket_where_clause(bucket_predicate)
|
|
1175
|
-
|
|
1223
|
+
source_row_hash_expr = self._row_hash_expr(value_columns)
|
|
1224
|
+
target_row_hash_expr = self._row_hash_expr(target_value_columns)
|
|
1176
1225
|
|
|
1177
1226
|
# row_numbers are Python ints derived from our own prior
|
|
1178
1227
|
# ROW_NUMBER() output (never user input) - safe to inline.
|
|
1179
1228
|
unique_row_numbers = sorted(set(int(n) for n in row_numbers))[: int(limit_samples)]
|
|
1180
1229
|
row_numbers_csv = ", ".join(str(n) for n in unique_row_numbers)
|
|
1181
1230
|
|
|
1182
|
-
def _numbered_query(fqtn: str) -> str:
|
|
1231
|
+
def _numbered_query(fqtn: str, order_by: str, select_list: str, row_hash_expr: str) -> str:
|
|
1183
1232
|
return f"""
|
|
1184
1233
|
SELECT row_number, {select_list}, {row_hash_expr} AS __row_hash
|
|
1185
1234
|
FROM (
|
|
@@ -1194,10 +1243,10 @@ class DatabricksConnector:
|
|
|
1194
1243
|
|
|
1195
1244
|
try:
|
|
1196
1245
|
source_rows = self._execute_to_dataframe(
|
|
1197
|
-
_numbered_query(src)
|
|
1246
|
+
_numbered_query(src, source_order_by, source_select_list, source_row_hash_expr)
|
|
1198
1247
|
).to_dict(orient="records")
|
|
1199
1248
|
target_rows = self._execute_to_dataframe(
|
|
1200
|
-
_numbered_query(tgt)
|
|
1249
|
+
_numbered_query(tgt, target_order_by, target_select_list, target_row_hash_expr)
|
|
1201
1250
|
).to_dict(orient="records")
|
|
1202
1251
|
except Exception as exc:
|
|
1203
1252
|
logger.exception(
|
|
@@ -1216,18 +1265,25 @@ class DatabricksConnector:
|
|
|
1216
1265
|
continue
|
|
1217
1266
|
|
|
1218
1267
|
mismatched_columns = [
|
|
1219
|
-
|
|
1220
|
-
if values_differ(src_row.get(
|
|
1268
|
+
source_to_target[src_col] for src_col in value_columns
|
|
1269
|
+
if values_differ(src_row.get(src_col), tgt_row.get(source_to_target[src_col]))
|
|
1221
1270
|
]
|
|
1222
1271
|
if not mismatched_columns:
|
|
1223
|
-
mismatched_columns = list(
|
|
1272
|
+
mismatched_columns = list(target_value_columns)
|
|
1224
1273
|
|
|
1225
1274
|
detail.append(
|
|
1226
1275
|
{
|
|
1227
1276
|
"key": {"row_number": src_row["row_number"]},
|
|
1228
1277
|
"mismatched_columns": mismatched_columns,
|
|
1229
|
-
"source_values": {
|
|
1230
|
-
|
|
1278
|
+
"source_values": {
|
|
1279
|
+
source_to_target[src_col]: src_row.get(src_col)
|
|
1280
|
+
for src_col in value_columns
|
|
1281
|
+
if source_to_target[src_col] in mismatched_columns
|
|
1282
|
+
},
|
|
1283
|
+
"target_values": {
|
|
1284
|
+
tgt_col: tgt_row.get(tgt_col)
|
|
1285
|
+
for tgt_col in mismatched_columns
|
|
1286
|
+
},
|
|
1231
1287
|
"source_row_hash": src_row.get("__row_hash"),
|
|
1232
1288
|
"target_row_hash": tgt_row.get("__row_hash"),
|
|
1233
1289
|
}
|
|
@@ -517,6 +517,30 @@ class CatalogValidationRequest(BaseModel):
|
|
|
517
517
|
),
|
|
518
518
|
)
|
|
519
519
|
|
|
520
|
+
column_map: Dict[str, str] = Field(
|
|
521
|
+
default_factory=dict,
|
|
522
|
+
description=(
|
|
523
|
+
"Optional map of source-table column name -> target-table "
|
|
524
|
+
"column name, for when an individual column was renamed "
|
|
525
|
+
"between source and target (e.g. source has 'cust_id', "
|
|
526
|
+
"target has 'customer_id'). An explicit pair like this is "
|
|
527
|
+
"treated as fully equivalent through the entire pipeline - "
|
|
528
|
+
"schema/type/nullable checks, null/distinct/min-max "
|
|
529
|
+
"statistics, whole-table fingerprint, row-hash diff, and "
|
|
530
|
+
"column-level mismatch detail - bypassing name-based column "
|
|
531
|
+
"matching entirely for that pair (unlike ignore_columns/"
|
|
532
|
+
"only_columns, which still require the name to appear in the "
|
|
533
|
+
"intersection). Unmapped columns are matched by identical "
|
|
534
|
+
"name as usual. Resolved before only_columns/ignore_columns/"
|
|
535
|
+
"ignore_datatype_columns apply, so those three act on the "
|
|
536
|
+
"resolved/canonical (target-side) name. A mapped name that "
|
|
537
|
+
"doesn't actually exist on either side produces a clear "
|
|
538
|
+
"error rather than a silent no-op. Known limitation: a "
|
|
539
|
+
"column configured as a primary key, or used as a Tier 3 "
|
|
540
|
+
"partition/bucket column, must not also appear here."
|
|
541
|
+
),
|
|
542
|
+
)
|
|
543
|
+
|
|
520
544
|
case_sensitive_columns: bool = Field(
|
|
521
545
|
default=False,
|
|
522
546
|
description="Case-sensitive column name comparison.",
|
|
@@ -758,6 +782,12 @@ class ColumnValidationResult(BaseModel):
|
|
|
758
782
|
column: str
|
|
759
783
|
status: ValidationStatus
|
|
760
784
|
|
|
785
|
+
# Populated only when column_map actually renamed this column (source
|
|
786
|
+
# name differs from the target/canonical name shown above) - mirrors
|
|
787
|
+
# TableValidationResult.source_table_name's "only set when it
|
|
788
|
+
# differs" convention.
|
|
789
|
+
source_column: Optional[str] = None
|
|
790
|
+
|
|
761
791
|
source_data_type: Optional[str] = None
|
|
762
792
|
target_data_type: Optional[str] = None
|
|
763
793
|
data_type_status: Optional[ValidationStatus] = None
|
|
@@ -799,6 +829,11 @@ class RowMismatchDetail(BaseModel):
|
|
|
799
829
|
primary_key: Dict[str, Any] = Field(default_factory=dict)
|
|
800
830
|
mismatch_column: str
|
|
801
831
|
|
|
832
|
+
# Populated only when column_map actually renamed this column -
|
|
833
|
+
# mismatch_column stays the canonical/target-side name (existing
|
|
834
|
+
# report rendering, existing tests, unaffected).
|
|
835
|
+
source_mismatch_column: Optional[str] = None
|
|
836
|
+
|
|
802
837
|
source_value: Optional[Any] = None
|
|
803
838
|
target_value: Optional[Any] = None
|
|
804
839
|
|