table-validator 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- table_validator/__init__.py +46 -0
- table_validator/auth/__init__.py +1 -0
- table_validator/auth/azure_auth.py +52 -0
- table_validator/auth/databricks_auth.py +31 -0
- table_validator/cli/__init__.py +1 -0
- table_validator/cli/main.py +722 -0
- table_validator/cli/partition_prompt.py +78 -0
- table_validator/cli/summary_table.py +146 -0
- table_validator/cli/wizard.py +429 -0
- table_validator/config/__init__.py +1 -0
- table_validator/config/manager.py +84 -0
- table_validator/config/schema.py +179 -0
- table_validator/connectors/__init__.py +1 -0
- table_validator/connectors/azure_connector.py +809 -0
- table_validator/connectors/databricks_connector.py +1230 -0
- table_validator/engine/__init__.py +1 -0
- table_validator/engine/comparison_engine.py +645 -0
- table_validator/models.py +952 -0
- table_validator/reports/__init__.py +1 -0
- table_validator/reports/excel_report.py +953 -0
- table_validator/validators/__init__.py +1 -0
- table_validator/validators/blob_discovery.py +467 -0
- table_validator/validators/catalog_validator.py +1863 -0
- table_validator/validators/row_validator.py +1727 -0
- table_validator-0.1.0.dist-info/METADATA +190 -0
- table_validator-0.1.0.dist-info/RECORD +30 -0
- table_validator-0.1.0.dist-info/WHEEL +5 -0
- table_validator-0.1.0.dist-info/entry_points.txt +2 -0
- table_validator-0.1.0.dist-info/licenses/LICENSE +21 -0
- table_validator-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,953 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Report Generator (CSV + Excel)
|
|
3
|
+
|
|
4
|
+
Consumes a CatalogValidationResponse (produced by
|
|
5
|
+
validators.catalog_validator.CatalogValidator) and renders it as either a flat .csv
|
|
6
|
+
file (Table Validation data only) or a formatted, multi-sheet .xlsx
|
|
7
|
+
workbook (Summary / Table Validation / Column Validation / Data
|
|
8
|
+
Mismatches / Row Hash Mismatches).
|
|
9
|
+
|
|
10
|
+
Deliberately kept out of comparison_engine.py per the project spec:
|
|
11
|
+
the validator returns clean structured data, and this module is the
|
|
12
|
+
only place that knows about report formatting (csv / openpyxl).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import csv
|
|
18
|
+
import logging
|
|
19
|
+
from typing import Any, List, Optional, Tuple
|
|
20
|
+
|
|
21
|
+
from openpyxl import Workbook
|
|
22
|
+
from openpyxl.styles import Alignment, Border, Font, PatternFill, Side
|
|
23
|
+
from openpyxl.utils import get_column_letter
|
|
24
|
+
from openpyxl.worksheet.worksheet import Worksheet
|
|
25
|
+
|
|
26
|
+
from table_validator.models import CatalogValidationResponse, ValidationStatus
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger(__name__)
|
|
29
|
+
|
|
30
|
+
# ---------------------------------------------------------------------------
|
|
31
|
+
# Shared column definitions
|
|
32
|
+
# ---------------------------------------------------------------------------
|
|
33
|
+
TABLE_HEADERS = [
|
|
34
|
+
"Source Schema",
|
|
35
|
+
"Source Table",
|
|
36
|
+
"Target Schema",
|
|
37
|
+
"Target Table",
|
|
38
|
+
"Overall Status",
|
|
39
|
+
"Schema Match",
|
|
40
|
+
"Column Order",
|
|
41
|
+
"Row Count (Src)",
|
|
42
|
+
"Row Count (Tgt)",
|
|
43
|
+
"Row Count Diff",
|
|
44
|
+
"Data Types",
|
|
45
|
+
"Nullable",
|
|
46
|
+
"Null Counts",
|
|
47
|
+
"Distinct Counts",
|
|
48
|
+
"Min/Max",
|
|
49
|
+
"Data Match",
|
|
50
|
+
"Mismatch Count",
|
|
51
|
+
"Mismatch %",
|
|
52
|
+
"Row Hash Mismatch Count",
|
|
53
|
+
"Row Hash Mismatch %",
|
|
54
|
+
"Tier Reached",
|
|
55
|
+
"Partition",
|
|
56
|
+
"Validation Timestamp",
|
|
57
|
+
"Duration",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
# 1-based column indices within TABLE_HEADERS holding a PASS/FAIL/ERROR/SKIPPED value.
|
|
61
|
+
_TABLE_STATUS_COLUMNS = {5, 6, 7, 11, 12, 13, 14, 15, 16}
|
|
62
|
+
|
|
63
|
+
# Which ValidationType (config/schema.py) owns each 0-based TABLE_HEADERS
|
|
64
|
+
# column, for hiding columns whose validation type wasn't selected to run
|
|
65
|
+
# (see _filter_table_columns). None = always shown regardless of
|
|
66
|
+
# enabled_validations. "Schema Match"/"Column Order" measure column-name/
|
|
67
|
+
# order agreement (COLUMN, despite the "Schema Match" label being a
|
|
68
|
+
# pre-existing naming quirk from before source types other than
|
|
69
|
+
# Databricks existed) - not schema/table existence, which has no
|
|
70
|
+
# dedicated column here (it's surfaced as separate MISSING_FROM_TARGET/
|
|
71
|
+
# EXTRA_IN_TARGET rows instead, unaffected by column filtering).
|
|
72
|
+
_TABLE_COLUMN_OWNERS = [
|
|
73
|
+
None, None, None, None, None, # Source/Target Schema/Table, Overall Status
|
|
74
|
+
"column", # Schema Match
|
|
75
|
+
"column", # Column Order
|
|
76
|
+
"row", "row", "row", # Row Count (Src/Tgt/Diff)
|
|
77
|
+
"column", # Data Types
|
|
78
|
+
"column", # Nullable
|
|
79
|
+
"column", # Null Counts
|
|
80
|
+
"column", # Distinct Counts
|
|
81
|
+
"column", # Min/Max
|
|
82
|
+
"row", # Data Match
|
|
83
|
+
"row", "row", # Mismatch Count/%
|
|
84
|
+
"row", "row", # Row Hash Mismatch Count/%
|
|
85
|
+
"row", # Tier Reached
|
|
86
|
+
"row", # Partition
|
|
87
|
+
None, None, # Validation Timestamp, Duration
|
|
88
|
+
]
|
|
89
|
+
|
|
90
|
+
COLUMN_HEADERS = [
|
|
91
|
+
"Source Schema", "Source Table", "Column", "Status",
|
|
92
|
+
"Source Type", "Target Type", "Type Status",
|
|
93
|
+
"Source Nullable", "Target Nullable", "Nullable Status",
|
|
94
|
+
"Source Nulls", "Target Nulls", "Null Status",
|
|
95
|
+
"Source Distinct", "Target Distinct", "Distinct Status",
|
|
96
|
+
"Source Min", "Source Max", "Target Min", "Target Max", "Min/Max Status",
|
|
97
|
+
"Error",
|
|
98
|
+
]
|
|
99
|
+
_COLUMN_STATUS_COLUMNS = {4, 7, 10, 13, 16, 21}
|
|
100
|
+
|
|
101
|
+
MISMATCH_HEADERS = [
|
|
102
|
+
"Source Schema", "Source Table", "Primary Key",
|
|
103
|
+
"Mismatch Column", "Source Value", "Target Value",
|
|
104
|
+
"Row Hash (Source)", "Row Hash (Target)", "Verified",
|
|
105
|
+
]
|
|
106
|
+
|
|
107
|
+
ROW_HASH_HEADERS = [
|
|
108
|
+
"Row #", "Source Schema", "Source Table",
|
|
109
|
+
"Primary Key", "Source Hash", "Target Hash", "Mismatch Status",
|
|
110
|
+
"Partition Bucket",
|
|
111
|
+
]
|
|
112
|
+
_ROW_HASH_STATUS_COLUMN = 7
|
|
113
|
+
|
|
114
|
+
SUGGESTION_HEADERS = [
|
|
115
|
+
"Source Schema", "Source Table", "Column", "Issue Type", "Suggestion",
|
|
116
|
+
]
|
|
117
|
+
|
|
118
|
+
SUMMARY_METRIC_LABELS = [
|
|
119
|
+
"Total Tables", "Passed Tables", "Failed Tables",
|
|
120
|
+
"Error Tables", "Skipped Tables", "Pass Percentage",
|
|
121
|
+
]
|
|
122
|
+
|
|
123
|
+
# ---------------------------------------------------------------------------
|
|
124
|
+
# Styling
|
|
125
|
+
# ---------------------------------------------------------------------------
|
|
126
|
+
FONT_NAME = "Arial"
|
|
127
|
+
HEADER_FILL = PatternFill("solid", fgColor="1F4E78")
|
|
128
|
+
HEADER_FONT = Font(name=FONT_NAME, size=11, bold=True, color="FFFFFF")
|
|
129
|
+
TITLE_FONT = Font(name=FONT_NAME, size=16, bold=True, color="1F4E78")
|
|
130
|
+
LABEL_FONT = Font(name=FONT_NAME, size=10, bold=True)
|
|
131
|
+
VALUE_FONT = Font(name=FONT_NAME, size=10)
|
|
132
|
+
THIN_BORDER = Border(*(Side(style="thin", color="D9D9D9"),) * 4)
|
|
133
|
+
|
|
134
|
+
STATUS_FILLS = {
|
|
135
|
+
ValidationStatus.PASS: PatternFill("solid", fgColor="C6EFCE"),
|
|
136
|
+
ValidationStatus.FAIL: PatternFill("solid", fgColor="FFC7CE"),
|
|
137
|
+
ValidationStatus.ERROR: PatternFill("solid", fgColor="FFD8A8"),
|
|
138
|
+
ValidationStatus.SKIPPED: PatternFill("solid", fgColor="E7E6E6"),
|
|
139
|
+
}
|
|
140
|
+
STATUS_FONTS = {
|
|
141
|
+
ValidationStatus.PASS: Font(name=FONT_NAME, size=10, bold=True, color="006100"),
|
|
142
|
+
ValidationStatus.FAIL: Font(name=FONT_NAME, size=10, bold=True, color="9C0006"),
|
|
143
|
+
ValidationStatus.ERROR: Font(name=FONT_NAME, size=10, bold=True, color="974706"),
|
|
144
|
+
ValidationStatus.SKIPPED: Font(name=FONT_NAME, size=10, bold=True, color="666666"),
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
# Sort rank so FAIL/ERROR surface above PASS/SKIPPED within a schema.
|
|
148
|
+
_STATUS_SORT_RANK = {
|
|
149
|
+
"FAIL": 0,
|
|
150
|
+
"ERROR": 0,
|
|
151
|
+
"MISSING_FROM_TARGET": 0,
|
|
152
|
+
"EXTRA_IN_TARGET": 0,
|
|
153
|
+
"PASS": 1,
|
|
154
|
+
"SKIPPED": 1,
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
# Row-hash mismatch statuses aren't ValidationStatus members - map them onto
|
|
158
|
+
# the same PASS/FAIL/ERROR/SKIPPED fill+font conventions for the Row Hash
|
|
159
|
+
# Mismatches sheet (all three outcomes here are a real difference -> FAIL).
|
|
160
|
+
_ROW_HASH_STATUS_FILL_MAP = {
|
|
161
|
+
"MISMATCH": ValidationStatus.FAIL,
|
|
162
|
+
"MISSING_IN_TARGET": ValidationStatus.FAIL,
|
|
163
|
+
"MISSING_IN_SOURCE": ValidationStatus.FAIL,
|
|
164
|
+
"DUPLICATE_KEY": ValidationStatus.FAIL,
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _status_value(status: Optional[ValidationStatus]) -> str:
|
|
169
|
+
if status is None:
|
|
170
|
+
return ""
|
|
171
|
+
return status.value if isinstance(status, ValidationStatus) else str(status)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _mismatch_count(table) -> Optional[int]:
|
|
175
|
+
if table.data is None:
|
|
176
|
+
return None
|
|
177
|
+
counts = [table.data.source_only_rows, table.data.target_only_rows, table.data.changed_rows]
|
|
178
|
+
if not all(c is None for c in counts):
|
|
179
|
+
return sum(c or 0 for c in counts)
|
|
180
|
+
# The tiered fail-fast funnel (Databricks-to-Databricks) never
|
|
181
|
+
# populates source_only_rows/target_only_rows/changed_rows - those
|
|
182
|
+
# only ever came from the legacy FULL-mode EXCEPT/hash-join path.
|
|
183
|
+
# Fall back to the row-hash comparison's own mismatch count, which is
|
|
184
|
+
# the real per-row mismatch figure for every table run through the
|
|
185
|
+
# tiered pipeline.
|
|
186
|
+
if table.data.row_hash_mismatch_count:
|
|
187
|
+
return table.data.row_hash_mismatch_count
|
|
188
|
+
return None
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _partition_summary(table) -> str:
|
|
192
|
+
if table.partitioned:
|
|
193
|
+
return (
|
|
194
|
+
f"{table.partition_column} "
|
|
195
|
+
f"({table.partition_buckets_culprit}/{table.partition_buckets_total} buckets differed)"
|
|
196
|
+
)
|
|
197
|
+
if table.partition_skip_reason:
|
|
198
|
+
return f"not partitioned ({table.partition_skip_reason})"
|
|
199
|
+
return ""
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _mismatch_pct(table, mismatch_count: Optional[int]) -> str:
|
|
203
|
+
if mismatch_count is None:
|
|
204
|
+
return ""
|
|
205
|
+
if table.data is not None and (
|
|
206
|
+
table.data.source_only_rows is None
|
|
207
|
+
and table.data.target_only_rows is None
|
|
208
|
+
and table.data.changed_rows is None
|
|
209
|
+
):
|
|
210
|
+
# Tiered-pipeline fallback (see _mismatch_count): reuse the
|
|
211
|
+
# row-hash comparison's own percentage, which is computed over
|
|
212
|
+
# the union of keys actually compared on either side - more
|
|
213
|
+
# accurate than dividing by row_count_source alone when keys are
|
|
214
|
+
# missing/extra on one side.
|
|
215
|
+
return f"{table.data.row_hash_mismatch_percentage:.2f}%"
|
|
216
|
+
total = table.row_count_source or 0
|
|
217
|
+
if total <= 0:
|
|
218
|
+
return ""
|
|
219
|
+
return f"{(mismatch_count / total) * 100:.2f}%"
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _filter_table_columns(
|
|
223
|
+
headers: List[str],
|
|
224
|
+
rows: List[List[Any]],
|
|
225
|
+
enabled_validations: Optional[set] = None,
|
|
226
|
+
) -> Tuple[List[str], List[List[Any]], set]:
|
|
227
|
+
"""
|
|
228
|
+
Drop TABLE_HEADERS columns (and the matching cell in every row) whose
|
|
229
|
+
owning validation type isn't in enabled_validations - e.g. hide "Data
|
|
230
|
+
Types"/"Null Counts"/etc. entirely when "column" wasn't selected to
|
|
231
|
+
run, rather than showing a column of meaningless SKIPPED values.
|
|
232
|
+
|
|
233
|
+
enabled_validations=None means "no filtering" (show everything),
|
|
234
|
+
matching the CSV/legacy behavior and any caller that doesn't know
|
|
235
|
+
about per-run validation selection.
|
|
236
|
+
|
|
237
|
+
Returns (filtered_headers, filtered_rows, status_columns) where
|
|
238
|
+
status_columns is the 1-based set of surviving columns that hold a
|
|
239
|
+
PASS/FAIL/ERROR/SKIPPED value, recomputed for the new column
|
|
240
|
+
positions (_TABLE_STATUS_COLUMNS' original indices no longer apply
|
|
241
|
+
once columns are dropped).
|
|
242
|
+
"""
|
|
243
|
+
if enabled_validations is None:
|
|
244
|
+
return headers, rows, _TABLE_STATUS_COLUMNS
|
|
245
|
+
|
|
246
|
+
keep_indices = [
|
|
247
|
+
i for i, owner in enumerate(_TABLE_COLUMN_OWNERS)
|
|
248
|
+
if owner is None or owner in enabled_validations
|
|
249
|
+
]
|
|
250
|
+
|
|
251
|
+
filtered_headers = [headers[i] for i in keep_indices]
|
|
252
|
+
filtered_rows = [[row[i] for i in keep_indices] for row in rows]
|
|
253
|
+
status_columns = {
|
|
254
|
+
new_idx + 1
|
|
255
|
+
for new_idx, old_idx in enumerate(keep_indices)
|
|
256
|
+
if (old_idx + 1) in _TABLE_STATUS_COLUMNS
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
return filtered_headers, filtered_rows, status_columns
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
# ---------------------------------------------------------------------------
|
|
263
|
+
# Row builders (shared by CSV + every Excel sheet)
|
|
264
|
+
# ---------------------------------------------------------------------------
|
|
265
|
+
def _build_table_rows(result: CatalogValidationResponse) -> List[List[Any]]:
|
|
266
|
+
timestamp = result.validation_timestamp or ""
|
|
267
|
+
duration = result.execution_time_seconds
|
|
268
|
+
|
|
269
|
+
rows: List[List[Any]] = []
|
|
270
|
+
for schema in result.schemas:
|
|
271
|
+
for table in schema.tables:
|
|
272
|
+
data_status = table.data.status if table.data else None
|
|
273
|
+
mismatch_count = _mismatch_count(table)
|
|
274
|
+
|
|
275
|
+
row_hash_count = table.data.row_hash_mismatch_count if table.data else 0
|
|
276
|
+
row_hash_pct = (
|
|
277
|
+
f"{table.data.row_hash_mismatch_percentage:.2f}%" if table.data else ""
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
rows.append(
|
|
281
|
+
[
|
|
282
|
+
schema.schema_name,
|
|
283
|
+
table.table,
|
|
284
|
+
schema.schema_name,
|
|
285
|
+
table.table,
|
|
286
|
+
_status_value(table.status),
|
|
287
|
+
_status_value(table.columns_status),
|
|
288
|
+
_status_value(table.column_order_status),
|
|
289
|
+
table.row_count_source,
|
|
290
|
+
table.row_count_target,
|
|
291
|
+
table.row_count_difference,
|
|
292
|
+
_status_value(table.data_types_status),
|
|
293
|
+
_status_value(table.nullable_status),
|
|
294
|
+
_status_value(table.null_counts_status),
|
|
295
|
+
_status_value(table.distinct_counts_status),
|
|
296
|
+
_status_value(table.min_max_status),
|
|
297
|
+
_status_value(data_status),
|
|
298
|
+
mismatch_count if mismatch_count is not None else "",
|
|
299
|
+
_mismatch_pct(table, mismatch_count),
|
|
300
|
+
row_hash_count,
|
|
301
|
+
row_hash_pct,
|
|
302
|
+
table.tier_reached.name,
|
|
303
|
+
_partition_summary(table),
|
|
304
|
+
timestamp,
|
|
305
|
+
duration,
|
|
306
|
+
]
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
for missing_table in schema.missing_tables:
|
|
310
|
+
rows.append(
|
|
311
|
+
[
|
|
312
|
+
schema.schema_name, missing_table, schema.schema_name, missing_table,
|
|
313
|
+
"MISSING_FROM_TARGET",
|
|
314
|
+
]
|
|
315
|
+
+ [""] * 17
|
|
316
|
+
+ [timestamp, duration]
|
|
317
|
+
)
|
|
318
|
+
|
|
319
|
+
for extra_table in schema.extra_tables:
|
|
320
|
+
rows.append(
|
|
321
|
+
[
|
|
322
|
+
schema.schema_name, extra_table, schema.schema_name, extra_table,
|
|
323
|
+
"EXTRA_IN_TARGET",
|
|
324
|
+
]
|
|
325
|
+
+ [""] * 17
|
|
326
|
+
+ [timestamp, duration]
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
# Sort: Source Schema asc, then FAIL/ERROR before PASS/SKIPPED, then Source Table asc.
|
|
330
|
+
rows.sort(key=lambda r: (r[0], _STATUS_SORT_RANK.get(r[4], 1), r[1]))
|
|
331
|
+
return rows
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _build_column_rows(result: CatalogValidationResponse) -> List[List[Any]]:
|
|
335
|
+
rows: List[List[Any]] = []
|
|
336
|
+
for schema in result.schemas:
|
|
337
|
+
for table in schema.tables:
|
|
338
|
+
for col in table.columns:
|
|
339
|
+
rows.append(
|
|
340
|
+
[
|
|
341
|
+
schema.schema_name, table.table, col.column, _status_value(col.status),
|
|
342
|
+
col.source_data_type, col.target_data_type, _status_value(col.data_type_status),
|
|
343
|
+
col.source_nullable, col.target_nullable, _status_value(col.nullable_status),
|
|
344
|
+
col.source_null_count, col.target_null_count, _status_value(col.null_count_status),
|
|
345
|
+
col.source_distinct_count, col.target_distinct_count, _status_value(col.distinct_count_status),
|
|
346
|
+
col.source_min, col.source_max, col.target_min, col.target_max,
|
|
347
|
+
_status_value(col.min_max_status),
|
|
348
|
+
col.error or "",
|
|
349
|
+
]
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
for missing_col in table.missing_columns:
|
|
353
|
+
rows.append(
|
|
354
|
+
[schema.schema_name, table.table, missing_col, "MISSING_FROM_TARGET"]
|
|
355
|
+
+ [""] * 18
|
|
356
|
+
)
|
|
357
|
+
for extra_col in table.extra_columns:
|
|
358
|
+
rows.append(
|
|
359
|
+
[schema.schema_name, table.table, extra_col, "EXTRA_IN_TARGET"]
|
|
360
|
+
+ [""] * 18
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
rows.sort(key=lambda r: (r[0], r[1], r[2]))
|
|
364
|
+
return rows
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _build_mismatch_rows(result: CatalogValidationResponse) -> List[List[Any]]:
|
|
368
|
+
rows: List[List[Any]] = []
|
|
369
|
+
for schema in result.schemas:
|
|
370
|
+
for table in schema.tables:
|
|
371
|
+
if table.data is None:
|
|
372
|
+
continue
|
|
373
|
+
for detail in table.data.sample_changed_detail:
|
|
374
|
+
key_text = ", ".join(f"{k}={v}" for k, v in detail.primary_key.items())
|
|
375
|
+
rows.append(
|
|
376
|
+
[
|
|
377
|
+
schema.schema_name,
|
|
378
|
+
table.table,
|
|
379
|
+
key_text,
|
|
380
|
+
detail.mismatch_column,
|
|
381
|
+
detail.source_value,
|
|
382
|
+
detail.target_value,
|
|
383
|
+
detail.source_row_hash,
|
|
384
|
+
detail.target_row_hash,
|
|
385
|
+
"Yes" if detail.verified else "No (row-number, unverified)",
|
|
386
|
+
]
|
|
387
|
+
)
|
|
388
|
+
|
|
389
|
+
# Group by Source Schema, Source Table, Primary Key.
|
|
390
|
+
rows.sort(key=lambda r: (r[0], r[1], r[2]))
|
|
391
|
+
return rows
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _build_row_hash_rows(result: CatalogValidationResponse) -> List[List[Any]]:
|
|
395
|
+
rows: List[List[Any]] = []
|
|
396
|
+
for schema in result.schemas:
|
|
397
|
+
for table in schema.tables:
|
|
398
|
+
if table.data is None:
|
|
399
|
+
continue
|
|
400
|
+
for mismatch in table.data.row_hash_mismatches:
|
|
401
|
+
rows.append(
|
|
402
|
+
[
|
|
403
|
+
schema.schema_name,
|
|
404
|
+
table.table,
|
|
405
|
+
mismatch.primary_key,
|
|
406
|
+
mismatch.source_hash,
|
|
407
|
+
mismatch.target_hash,
|
|
408
|
+
mismatch.status,
|
|
409
|
+
mismatch.partition_bucket or "",
|
|
410
|
+
]
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
# Group by Source Schema, Source Table, Primary Key.
|
|
414
|
+
rows.sort(key=lambda r: (r[0], r[1], r[2]))
|
|
415
|
+
|
|
416
|
+
# Prepend a 1-based sequential row number, assigned after sorting so it
|
|
417
|
+
# reflects final display order (not discovery order).
|
|
418
|
+
return [[i, *row] for i, row in enumerate(rows, start=1)]
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _build_suggestion_rows(result: CatalogValidationResponse) -> List[List[Any]]:
|
|
422
|
+
"""
|
|
423
|
+
One plain-English sentence per issue found on a table, covering every
|
|
424
|
+
category that can independently fail a table's Overall Status:
|
|
425
|
+
schema/constraint issues (missing/extra column, column order, data
|
|
426
|
+
type, nullable), per-column statistics (null count, distinct count,
|
|
427
|
+
min/max), row count, and row-level data (pointing at the Data
|
|
428
|
+
Mismatches / Row Hash Mismatches sheets for full detail rather than
|
|
429
|
+
duplicating it here).
|
|
430
|
+
|
|
431
|
+
If a table's row data actually matches (row count, row-hash/data
|
|
432
|
+
comparison all passed) but the table still failed purely on
|
|
433
|
+
schema/constraint or statistics grounds, an extra summary sentence
|
|
434
|
+
calls that out explicitly, since it's an easy thing to miss buried in
|
|
435
|
+
status columns. Every FAILed table gets at least one row here - if
|
|
436
|
+
none of the specific categories below apply, a fallback sentence
|
|
437
|
+
still points at Table Validation for the table's own Error field.
|
|
438
|
+
"""
|
|
439
|
+
rows: List[List[Any]] = []
|
|
440
|
+
|
|
441
|
+
for schema in result.schemas:
|
|
442
|
+
for table in schema.tables:
|
|
443
|
+
if table.status not in (ValidationStatus.FAIL, ValidationStatus.ERROR):
|
|
444
|
+
continue
|
|
445
|
+
|
|
446
|
+
table_issues: List[List[Any]] = []
|
|
447
|
+
|
|
448
|
+
if table.schema_blocking:
|
|
449
|
+
# BLOCKING schema difference (Tier 0) - the table was
|
|
450
|
+
# aborted before any row-level tier ran, so there is no
|
|
451
|
+
# row-hash/data finding to report alongside this. Surface
|
|
452
|
+
# that explicitly rather than falling through to the
|
|
453
|
+
# generic per-category checks below, most of which won't
|
|
454
|
+
# have fired for a blocked table anyway.
|
|
455
|
+
table_issues.append([
|
|
456
|
+
schema.schema_name, table.table, "-", "Blocked at Schema Check",
|
|
457
|
+
f"Table '{table.table}' has a schema difference severe enough "
|
|
458
|
+
f"(missing/extra column, incompatible data type, or a missing "
|
|
459
|
+
f"configured key column) that row-level comparison was not "
|
|
460
|
+
f"attempted - resolve the schema issue(s) below first, then "
|
|
461
|
+
f"re-run to check the data.",
|
|
462
|
+
])
|
|
463
|
+
|
|
464
|
+
for col in table.missing_columns:
|
|
465
|
+
table_issues.append([
|
|
466
|
+
schema.schema_name, table.table, col, "Missing Column",
|
|
467
|
+
f"Column '{col}' exists in the source but not in the target - "
|
|
468
|
+
f"it may not have been migrated, or was renamed/dropped.",
|
|
469
|
+
])
|
|
470
|
+
|
|
471
|
+
for col in table.extra_columns:
|
|
472
|
+
table_issues.append([
|
|
473
|
+
schema.schema_name, table.table, col, "Extra Column",
|
|
474
|
+
f"Column '{col}' exists in the target but not in the source - "
|
|
475
|
+
f"check whether it was added intentionally or is leftover from a prior load.",
|
|
476
|
+
])
|
|
477
|
+
|
|
478
|
+
if table.column_order_status == ValidationStatus.FAIL:
|
|
479
|
+
table_issues.append([
|
|
480
|
+
schema.schema_name, table.table, "-", "Column Order",
|
|
481
|
+
f"Columns match by name but are in a different order between source "
|
|
482
|
+
f"({', '.join(table.source_column_order)}) and target "
|
|
483
|
+
f"({', '.join(table.target_column_order)}).",
|
|
484
|
+
])
|
|
485
|
+
|
|
486
|
+
if table.row_count_status == ValidationStatus.FAIL:
|
|
487
|
+
diff = table.row_count_difference
|
|
488
|
+
direction = "more" if (diff or 0) > 0 else "fewer"
|
|
489
|
+
table_issues.append([
|
|
490
|
+
schema.schema_name, table.table, "-", "Row Count Mismatch",
|
|
491
|
+
f"Target has {abs(diff) if diff is not None else 'a different number of'} "
|
|
492
|
+
f"{direction} rows than source ({table.row_count_source} vs "
|
|
493
|
+
f"{table.row_count_target}) - check for a partial load, duplicate rows, "
|
|
494
|
+
f"or rows deleted/inserted after migration.",
|
|
495
|
+
])
|
|
496
|
+
elif table.row_count_status == ValidationStatus.ERROR:
|
|
497
|
+
table_issues.append([
|
|
498
|
+
schema.schema_name, table.table, "-", "Row Count Error",
|
|
499
|
+
f"Row count could not be verified for this table"
|
|
500
|
+
+ (f": {table.error}" if table.error else " - see the Error column in Table Validation.")
|
|
501
|
+
])
|
|
502
|
+
|
|
503
|
+
for col in table.columns:
|
|
504
|
+
if col.data_type_status == ValidationStatus.FAIL:
|
|
505
|
+
table_issues.append([
|
|
506
|
+
schema.schema_name, table.table, col.column, "Data Type Mismatch",
|
|
507
|
+
f"Column '{col.column}' is {col.source_data_type} in the source but "
|
|
508
|
+
f"{col.target_data_type} in the target - if the row data still matches, "
|
|
509
|
+
f"only the declared type differs; if not, check for precision/format loss "
|
|
510
|
+
f"during migration.",
|
|
511
|
+
])
|
|
512
|
+
|
|
513
|
+
if col.nullable_status == ValidationStatus.FAIL:
|
|
514
|
+
src_null = "nullable" if col.source_nullable else "NOT NULL"
|
|
515
|
+
tgt_null = "nullable" if col.target_nullable else "NOT NULL"
|
|
516
|
+
table_issues.append([
|
|
517
|
+
schema.schema_name, table.table, col.column, "Nullable Mismatch",
|
|
518
|
+
f"Column '{col.column}' is declared {src_null} in the source but "
|
|
519
|
+
f"{tgt_null} in the target - this is a constraint difference, not "
|
|
520
|
+
f"necessarily a data problem, unless the stricter side is expected "
|
|
521
|
+
f"to reject values the other side allows.",
|
|
522
|
+
])
|
|
523
|
+
|
|
524
|
+
if col.null_count_status == ValidationStatus.FAIL:
|
|
525
|
+
table_issues.append([
|
|
526
|
+
schema.schema_name, table.table, col.column, "Null Count Mismatch",
|
|
527
|
+
f"Column '{col.column}' has {col.source_null_count} NULLs in the source "
|
|
528
|
+
f"but {col.target_null_count} in the target - some rows likely gained or "
|
|
529
|
+
f"lost a NULL value for this column during migration.",
|
|
530
|
+
])
|
|
531
|
+
|
|
532
|
+
if col.distinct_count_status == ValidationStatus.FAIL:
|
|
533
|
+
table_issues.append([
|
|
534
|
+
schema.schema_name, table.table, col.column, "Distinct Count Mismatch",
|
|
535
|
+
f"Column '{col.column}' has {col.source_distinct_count} distinct values "
|
|
536
|
+
f"in the source but {col.target_distinct_count} in the target - check for "
|
|
537
|
+
f"duplicate or collapsed values, or rows missing on one side.",
|
|
538
|
+
])
|
|
539
|
+
|
|
540
|
+
if col.min_max_status == ValidationStatus.FAIL:
|
|
541
|
+
table_issues.append([
|
|
542
|
+
schema.schema_name, table.table, col.column, "Min/Max Mismatch",
|
|
543
|
+
f"Column '{col.column}' ranges from {col.source_min} to {col.source_max} "
|
|
544
|
+
f"in the source but {col.target_min} to {col.target_max} in the target - "
|
|
545
|
+
f"check for outlier rows unique to one side, or a truncated/extended value range.",
|
|
546
|
+
])
|
|
547
|
+
|
|
548
|
+
data = table.data
|
|
549
|
+
if data is not None:
|
|
550
|
+
if data.row_hash_mismatch_count and data.row_hash_mismatch_count > 0:
|
|
551
|
+
by_row_number = data.key_columns == ["row_number"]
|
|
552
|
+
by_clause = (
|
|
553
|
+
"by row position (best-effort - no primary key "
|
|
554
|
+
"configured, so this cannot guarantee the same "
|
|
555
|
+
"record on both sides)"
|
|
556
|
+
if by_row_number
|
|
557
|
+
else "by primary key"
|
|
558
|
+
)
|
|
559
|
+
detail_clause = (
|
|
560
|
+
"values (marked unverified for row-position matches), if available"
|
|
561
|
+
if by_row_number
|
|
562
|
+
else "values, if available"
|
|
563
|
+
)
|
|
564
|
+
table_issues.append([
|
|
565
|
+
schema.schema_name, table.table, "-", "Row Data Mismatch",
|
|
566
|
+
f"{data.row_hash_mismatch_count} row(s) "
|
|
567
|
+
f"({data.row_hash_mismatch_percentage:.2f}%) differ between source and "
|
|
568
|
+
f"target {by_clause} - see the 'Row Hash Mismatches' sheet for the "
|
|
569
|
+
f"affected keys, and 'Data Mismatches' for the specific column(s) and "
|
|
570
|
+
f"{detail_clause}.",
|
|
571
|
+
])
|
|
572
|
+
elif data.status == ValidationStatus.FAIL and (
|
|
573
|
+
(data.source_only_rows or 0) > 0 or (data.target_only_rows or 0) > 0
|
|
574
|
+
or (data.changed_rows or 0) > 0
|
|
575
|
+
):
|
|
576
|
+
table_issues.append([
|
|
577
|
+
schema.schema_name, table.table, "-", "Row Data Mismatch",
|
|
578
|
+
f"Row-level comparison found {data.source_only_rows or 0} row(s) only in "
|
|
579
|
+
f"the source, {data.target_only_rows or 0} only in the target, and "
|
|
580
|
+
f"{data.changed_rows or 0} changed - see the 'Data Mismatches' sheet for detail.",
|
|
581
|
+
])
|
|
582
|
+
elif data.status == ValidationStatus.ERROR:
|
|
583
|
+
table_issues.append([
|
|
584
|
+
schema.schema_name, table.table, "-", "Row Data Comparison Error",
|
|
585
|
+
f"Row-level comparison could not complete for this table"
|
|
586
|
+
+ (f": {data.error}" if data.error else " - see the Error column in Table Validation.")
|
|
587
|
+
])
|
|
588
|
+
|
|
589
|
+
if not table_issues:
|
|
590
|
+
# Table FAILed/ERRORed but none of the categories above
|
|
591
|
+
# explain why (e.g. an ERROR raised before any stage ran) -
|
|
592
|
+
# never leave a failed table unexplained.
|
|
593
|
+
table_issues.append([
|
|
594
|
+
schema.schema_name, table.table, "-", "Unclassified",
|
|
595
|
+
f"Table '{table.table}' has status {table.status.value} but the reason "
|
|
596
|
+
f"doesn't match a known category here - check the Error column and "
|
|
597
|
+
f"per-check status columns on the 'Table Validation' sheet for this table.",
|
|
598
|
+
])
|
|
599
|
+
|
|
600
|
+
schema_only_issue = (
|
|
601
|
+
not table.schema_blocking
|
|
602
|
+
and table.status == ValidationStatus.FAIL
|
|
603
|
+
and table.row_count_status != ValidationStatus.FAIL
|
|
604
|
+
and (table.data is None or table.data.status != ValidationStatus.FAIL)
|
|
605
|
+
and not any(issue[3] == "Row Data Mismatch" for issue in table_issues)
|
|
606
|
+
)
|
|
607
|
+
if schema_only_issue:
|
|
608
|
+
rows.append([
|
|
609
|
+
schema.schema_name, table.table, "-", "Summary",
|
|
610
|
+
f"Table '{table.table}' failed, but the row data matches "
|
|
611
|
+
f"(row counts and row-hash/data comparison passed) - the failure "
|
|
612
|
+
f"is caused entirely by the schema/constraint issue(s) below.",
|
|
613
|
+
])
|
|
614
|
+
rows.extend(table_issues)
|
|
615
|
+
|
|
616
|
+
rows.sort(key=lambda r: (r[0], r[1]))
|
|
617
|
+
return rows
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def _build_summary_metrics(result: CatalogValidationResponse) -> List[Tuple[str, Any]]:
|
|
621
|
+
total = passed = failed = errors = skipped = 0
|
|
622
|
+
for schema in result.schemas:
|
|
623
|
+
for table in schema.tables:
|
|
624
|
+
total += 1
|
|
625
|
+
if table.status == ValidationStatus.PASS:
|
|
626
|
+
passed += 1
|
|
627
|
+
elif table.status == ValidationStatus.ERROR:
|
|
628
|
+
errors += 1
|
|
629
|
+
elif table.status == ValidationStatus.SKIPPED:
|
|
630
|
+
skipped += 1
|
|
631
|
+
else:
|
|
632
|
+
failed += 1
|
|
633
|
+
total += len(schema.missing_tables)
|
|
634
|
+
failed += len(schema.missing_tables)
|
|
635
|
+
|
|
636
|
+
pass_pct = f"{(passed / total) * 100:.2f}%" if total else "0.00%"
|
|
637
|
+
|
|
638
|
+
return [
|
|
639
|
+
("Total Tables", total),
|
|
640
|
+
("Passed Tables", passed),
|
|
641
|
+
("Failed Tables", failed),
|
|
642
|
+
("Error Tables", errors),
|
|
643
|
+
("Skipped Tables", skipped),
|
|
644
|
+
("Pass Percentage", pass_pct),
|
|
645
|
+
]
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
# ---------------------------------------------------------------------------
|
|
649
|
+
# CSV (Table Validation data only)
|
|
650
|
+
# ---------------------------------------------------------------------------
|
|
651
|
+
def generate_csv_report(
|
|
652
|
+
result: CatalogValidationResponse,
|
|
653
|
+
output_path: str,
|
|
654
|
+
) -> str:
|
|
655
|
+
"""
|
|
656
|
+
Render the per-table validation results of a CatalogValidationResponse
|
|
657
|
+
as a flat .csv file. Returns the output_path for convenience.
|
|
658
|
+
"""
|
|
659
|
+
logger.info(
|
|
660
|
+
"Generating CSV report | source=%s | target=%s | -> %s",
|
|
661
|
+
result.source_catalog, result.target_catalog, output_path,
|
|
662
|
+
)
|
|
663
|
+
|
|
664
|
+
with open(output_path, "w", newline="", encoding="utf-8-sig") as f:
|
|
665
|
+
writer = csv.writer(f)
|
|
666
|
+
writer.writerow(TABLE_HEADERS)
|
|
667
|
+
writer.writerows(_build_table_rows(result))
|
|
668
|
+
|
|
669
|
+
logger.info("CSV report written to %s", output_path)
|
|
670
|
+
return output_path
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
# ---------------------------------------------------------------------------
|
|
674
|
+
# Excel helpers
|
|
675
|
+
# ---------------------------------------------------------------------------
|
|
676
|
+
def _write_header_row(ws: Worksheet, headers: List[str]) -> None:
|
|
677
|
+
for col_idx, header in enumerate(headers, start=1):
|
|
678
|
+
cell = ws.cell(row=1, column=col_idx, value=header)
|
|
679
|
+
cell.font = HEADER_FONT
|
|
680
|
+
cell.fill = HEADER_FILL
|
|
681
|
+
cell.alignment = Alignment(horizontal="center", vertical="center", wrap_text=True)
|
|
682
|
+
cell.border = THIN_BORDER
|
|
683
|
+
ws.freeze_panes = ws.cell(row=2, column=1)
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
def _write_rows(
|
|
687
|
+
ws: Worksheet,
|
|
688
|
+
rows: List[List[Any]],
|
|
689
|
+
status_columns: set,
|
|
690
|
+
group_col: Optional[int] = None,
|
|
691
|
+
custom_status_map: Optional[dict] = None,
|
|
692
|
+
) -> int:
|
|
693
|
+
"""
|
|
694
|
+
Write rows starting at row 2. If group_col is set, consecutive rows
|
|
695
|
+
sharing that column's value are put into a collapsible outline group
|
|
696
|
+
(Excel row grouping) - one level, matching the schema boundary.
|
|
697
|
+
Returns the last row index written (1 if no data rows).
|
|
698
|
+
"""
|
|
699
|
+
row_idx = 2
|
|
700
|
+
prev_group_value = None
|
|
701
|
+
group_start = None
|
|
702
|
+
|
|
703
|
+
def _close_group(end_row: int) -> None:
|
|
704
|
+
if group_start is not None and end_row > group_start:
|
|
705
|
+
for r in range(group_start, end_row):
|
|
706
|
+
ws.row_dimensions[r].outline_level = 1
|
|
707
|
+
|
|
708
|
+
for values in rows:
|
|
709
|
+
if group_col is not None:
|
|
710
|
+
current_value = values[group_col - 1]
|
|
711
|
+
if current_value != prev_group_value:
|
|
712
|
+
if prev_group_value is not None:
|
|
713
|
+
_close_group(row_idx)
|
|
714
|
+
prev_group_value = current_value
|
|
715
|
+
group_start = row_idx + 1 # first detail row after the group's own header row
|
|
716
|
+
|
|
717
|
+
for col_idx, value in enumerate(values, start=1):
|
|
718
|
+
cell = ws.cell(row=row_idx, column=col_idx, value=value)
|
|
719
|
+
cell.font = VALUE_FONT
|
|
720
|
+
cell.border = THIN_BORDER
|
|
721
|
+
cell.alignment = Alignment(vertical="center", wrap_text=False)
|
|
722
|
+
|
|
723
|
+
if col_idx in status_columns:
|
|
724
|
+
status = next((s for s in ValidationStatus if s.value == value), None)
|
|
725
|
+
if status is None and custom_status_map is not None:
|
|
726
|
+
status = custom_status_map.get(value)
|
|
727
|
+
if status in STATUS_FILLS:
|
|
728
|
+
cell.fill = STATUS_FILLS[status]
|
|
729
|
+
cell.font = STATUS_FONTS[status]
|
|
730
|
+
elif value in ("MISSING_FROM_TARGET", "EXTRA_IN_TARGET"):
|
|
731
|
+
cell.fill = STATUS_FILLS[ValidationStatus.FAIL]
|
|
732
|
+
cell.font = STATUS_FONTS[ValidationStatus.FAIL]
|
|
733
|
+
|
|
734
|
+
row_idx += 1
|
|
735
|
+
|
|
736
|
+
if group_col is not None:
|
|
737
|
+
_close_group(row_idx)
|
|
738
|
+
|
|
739
|
+
return row_idx - 1
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def _autofit(ws: Worksheet, headers: List[str], rows: List[List[Any]]) -> None:
|
|
743
|
+
for col_idx, header in enumerate(headers, start=1):
|
|
744
|
+
max_len = len(str(header))
|
|
745
|
+
for row_values in rows:
|
|
746
|
+
value = row_values[col_idx - 1]
|
|
747
|
+
if value is not None:
|
|
748
|
+
max_len = max(max_len, len(str(value)))
|
|
749
|
+
ws.column_dimensions[get_column_letter(col_idx)].width = min(max_len + 2, 60)
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
def _enable_filter(ws: Worksheet, num_cols: int, last_row: int) -> None:
|
|
753
|
+
ws.auto_filter.ref = f"A1:{get_column_letter(num_cols)}{max(last_row, 1)}"
|
|
754
|
+
|
|
755
|
+
|
|
756
|
+
# ---------------------------------------------------------------------------
|
|
757
|
+
# Excel sheet builders
|
|
758
|
+
# ---------------------------------------------------------------------------
|
|
759
|
+
def _build_summary_sheet(
|
|
760
|
+
wb: Workbook,
|
|
761
|
+
result: CatalogValidationResponse,
|
|
762
|
+
source_type: Optional[str] = None,
|
|
763
|
+
validations_run: Optional[str] = None,
|
|
764
|
+
) -> None:
|
|
765
|
+
ws = wb.active
|
|
766
|
+
ws.title = "Summary"
|
|
767
|
+
|
|
768
|
+
ws["A1"] = "Databricks Catalog Validation Report"
|
|
769
|
+
ws["A1"].font = TITLE_FONT
|
|
770
|
+
ws.merge_cells("A1:B1")
|
|
771
|
+
|
|
772
|
+
ws["A2"] = f"Source: {result.source_catalog} -> Target: {result.target_catalog}"
|
|
773
|
+
ws["A2"].font = VALUE_FONT
|
|
774
|
+
ws.merge_cells("A2:B2")
|
|
775
|
+
|
|
776
|
+
row = 4
|
|
777
|
+
fields = [("Overall Status", _status_value(result.status))]
|
|
778
|
+
if source_type:
|
|
779
|
+
fields.append(("Source Type", source_type))
|
|
780
|
+
if validations_run:
|
|
781
|
+
fields.append(("Validations Run", validations_run))
|
|
782
|
+
fields += [
|
|
783
|
+
("Validation Timestamp", result.validation_timestamp or ""),
|
|
784
|
+
("Duration (s)", result.execution_time_seconds),
|
|
785
|
+
]
|
|
786
|
+
for label, value in fields:
|
|
787
|
+
ws.cell(row=row, column=1, value=label).font = LABEL_FONT
|
|
788
|
+
cell = ws.cell(row=row, column=2, value=value)
|
|
789
|
+
cell.font = VALUE_FONT
|
|
790
|
+
if label == "Overall Status":
|
|
791
|
+
status = next((s for s in ValidationStatus if s.value == value), None)
|
|
792
|
+
if status in STATUS_FILLS:
|
|
793
|
+
cell.fill = STATUS_FILLS[status]
|
|
794
|
+
cell.font = STATUS_FONTS[status]
|
|
795
|
+
row += 1
|
|
796
|
+
|
|
797
|
+
row += 1
|
|
798
|
+
ws.cell(row=row, column=1, value="Table Summary").font = Font(
|
|
799
|
+
name=FONT_NAME, size=12, bold=True, color="1F4E78"
|
|
800
|
+
)
|
|
801
|
+
row += 1
|
|
802
|
+
|
|
803
|
+
header_row = row
|
|
804
|
+
ws.cell(row=header_row, column=1, value="Metric").font = HEADER_FONT
|
|
805
|
+
ws.cell(row=header_row, column=1).fill = HEADER_FILL
|
|
806
|
+
ws.cell(row=header_row, column=2, value="Value").font = HEADER_FONT
|
|
807
|
+
ws.cell(row=header_row, column=2).fill = HEADER_FILL
|
|
808
|
+
row += 1
|
|
809
|
+
|
|
810
|
+
for label, value in _build_summary_metrics(result):
|
|
811
|
+
ws.cell(row=row, column=1, value=label).font = VALUE_FONT
|
|
812
|
+
cell = ws.cell(row=row, column=2, value=value)
|
|
813
|
+
cell.font = VALUE_FONT
|
|
814
|
+
cell.alignment = Alignment(horizontal="right")
|
|
815
|
+
row += 1
|
|
816
|
+
|
|
817
|
+
ws.column_dimensions["A"].width = 28
|
|
818
|
+
ws.column_dimensions["B"].width = 40
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
def _build_table_validation_sheet(
|
|
822
|
+
wb: Workbook,
|
|
823
|
+
result: CatalogValidationResponse,
|
|
824
|
+
enabled_validations: Optional[set] = None,
|
|
825
|
+
) -> None:
|
|
826
|
+
ws = wb.create_sheet("Table Validation")
|
|
827
|
+
rows = _build_table_rows(result)
|
|
828
|
+
headers, rows, status_columns = _filter_table_columns(
|
|
829
|
+
TABLE_HEADERS, rows, enabled_validations
|
|
830
|
+
)
|
|
831
|
+
|
|
832
|
+
_write_header_row(ws, headers)
|
|
833
|
+
last_row = _write_rows(ws, rows, status_columns, group_col=1)
|
|
834
|
+
_autofit(ws, headers, rows)
|
|
835
|
+
_enable_filter(ws, len(headers), last_row)
|
|
836
|
+
|
|
837
|
+
|
|
838
|
+
def _build_column_validation_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
|
|
839
|
+
ws = wb.create_sheet("Column Validation")
|
|
840
|
+
rows = _build_column_rows(result)
|
|
841
|
+
|
|
842
|
+
_write_header_row(ws, COLUMN_HEADERS)
|
|
843
|
+
last_row = _write_rows(ws, rows, _COLUMN_STATUS_COLUMNS, group_col=1)
|
|
844
|
+
_autofit(ws, COLUMN_HEADERS, rows)
|
|
845
|
+
_enable_filter(ws, len(COLUMN_HEADERS), last_row)
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
def _build_data_mismatches_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
|
|
849
|
+
ws = wb.create_sheet("Data Mismatches")
|
|
850
|
+
rows = _build_mismatch_rows(result)
|
|
851
|
+
|
|
852
|
+
_write_header_row(ws, MISMATCH_HEADERS)
|
|
853
|
+
last_row = _write_rows(ws, rows, set(), group_col=1)
|
|
854
|
+
_autofit(ws, MISMATCH_HEADERS, rows)
|
|
855
|
+
_enable_filter(ws, len(MISMATCH_HEADERS), last_row)
|
|
856
|
+
|
|
857
|
+
|
|
858
|
+
def _build_row_hash_mismatches_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
|
|
859
|
+
ws = wb.create_sheet("Row Hash Mismatches")
|
|
860
|
+
rows = _build_row_hash_rows(result)
|
|
861
|
+
|
|
862
|
+
_write_header_row(ws, ROW_HASH_HEADERS)
|
|
863
|
+
last_row = _write_rows(
|
|
864
|
+
ws, rows, {_ROW_HASH_STATUS_COLUMN}, group_col=2,
|
|
865
|
+
custom_status_map=_ROW_HASH_STATUS_FILL_MAP,
|
|
866
|
+
)
|
|
867
|
+
_autofit(ws, ROW_HASH_HEADERS, rows)
|
|
868
|
+
_enable_filter(ws, len(ROW_HASH_HEADERS), last_row)
|
|
869
|
+
|
|
870
|
+
|
|
871
|
+
def _build_suggestions_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
|
|
872
|
+
ws = wb.create_sheet("Suggestions")
|
|
873
|
+
rows = _build_suggestion_rows(result)
|
|
874
|
+
|
|
875
|
+
_write_header_row(ws, SUGGESTION_HEADERS)
|
|
876
|
+
|
|
877
|
+
row_idx = 2
|
|
878
|
+
for values in rows:
|
|
879
|
+
for col_idx, value in enumerate(values, start=1):
|
|
880
|
+
cell = ws.cell(row=row_idx, column=col_idx, value=value)
|
|
881
|
+
cell.font = VALUE_FONT
|
|
882
|
+
cell.border = THIN_BORDER
|
|
883
|
+
cell.alignment = Alignment(vertical="top", wrap_text=(col_idx == len(SUGGESTION_HEADERS)))
|
|
884
|
+
row_idx += 1
|
|
885
|
+
|
|
886
|
+
ws.column_dimensions[get_column_letter(1)].width = 18
|
|
887
|
+
ws.column_dimensions[get_column_letter(2)].width = 20
|
|
888
|
+
ws.column_dimensions[get_column_letter(3)].width = 16
|
|
889
|
+
ws.column_dimensions[get_column_letter(4)].width = 18
|
|
890
|
+
ws.column_dimensions[get_column_letter(5)].width = 90
|
|
891
|
+
|
|
892
|
+
ws.freeze_panes = ws.cell(row=2, column=1)
|
|
893
|
+
ws.auto_filter.ref = f"A1:{get_column_letter(len(SUGGESTION_HEADERS))}{max(row_idx - 1, 1)}"
|
|
894
|
+
|
|
895
|
+
|
|
896
|
+
# ---------------------------------------------------------------------------
|
|
897
|
+
# Public entry point
|
|
898
|
+
# ---------------------------------------------------------------------------
|
|
899
|
+
def generate_excel_report(
|
|
900
|
+
result: CatalogValidationResponse,
|
|
901
|
+
output_path: str,
|
|
902
|
+
source_type: Optional[str] = None,
|
|
903
|
+
enabled_validations: Optional[set] = None,
|
|
904
|
+
) -> str:
|
|
905
|
+
"""
|
|
906
|
+
Render a CatalogValidationResponse as a formatted, multi-sheet .xlsx
|
|
907
|
+
workbook: Summary, Table Validation, Column Validation, Data
|
|
908
|
+
Mismatches, Row Hash Mismatches, Suggestions. Returns the output_path
|
|
909
|
+
for convenience.
|
|
910
|
+
|
|
911
|
+
source_type (e.g. "databricks"/"azure_blob"/"azure_sql") is optional
|
|
912
|
+
and purely cosmetic - shown on the Summary sheet next to Overall
|
|
913
|
+
Status so it's clear at a glance what kind of source was compared,
|
|
914
|
+
since the report format itself is identical regardless of source.
|
|
915
|
+
|
|
916
|
+
enabled_validations (e.g. {"catalog", "row"}) is optional; when given,
|
|
917
|
+
Table Validation columns and whole sheets belonging to a validation
|
|
918
|
+
type NOT in this set are omitted entirely - "column" gates the
|
|
919
|
+
Column Validation sheet plus the column-level columns on Table
|
|
920
|
+
Validation, "row" gates Data Mismatches/Row Hash Mismatches plus the
|
|
921
|
+
row-level columns on Table Validation. None means "show everything"
|
|
922
|
+
(no filtering), matching prior behavior for any caller that doesn't
|
|
923
|
+
pass it.
|
|
924
|
+
"""
|
|
925
|
+
logger.info(
|
|
926
|
+
"Generating Excel report | source=%s | target=%s | source_type=%s | "
|
|
927
|
+
"enabled_validations=%s | -> %s",
|
|
928
|
+
result.source_catalog, result.target_catalog, source_type,
|
|
929
|
+
enabled_validations, output_path,
|
|
930
|
+
)
|
|
931
|
+
|
|
932
|
+
validations_run = (
|
|
933
|
+
", ".join(sorted(enabled_validations)) if enabled_validations is not None else None
|
|
934
|
+
)
|
|
935
|
+
|
|
936
|
+
wb = Workbook()
|
|
937
|
+
|
|
938
|
+
_build_summary_sheet(wb, result, source_type, validations_run)
|
|
939
|
+
_build_table_validation_sheet(wb, result, enabled_validations)
|
|
940
|
+
|
|
941
|
+
if enabled_validations is None or "column" in enabled_validations:
|
|
942
|
+
_build_column_validation_sheet(wb, result)
|
|
943
|
+
|
|
944
|
+
if enabled_validations is None or "row" in enabled_validations:
|
|
945
|
+
_build_data_mismatches_sheet(wb, result)
|
|
946
|
+
_build_row_hash_mismatches_sheet(wb, result)
|
|
947
|
+
|
|
948
|
+
_build_suggestions_sheet(wb, result)
|
|
949
|
+
|
|
950
|
+
wb.save(output_path)
|
|
951
|
+
|
|
952
|
+
logger.info("Excel report written to %s", output_path)
|
|
953
|
+
return output_path
|