table-validator 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,953 @@
1
+ """
2
+ Report Generator (CSV + Excel)
3
+
4
+ Consumes a CatalogValidationResponse (produced by
5
+ validators.catalog_validator.CatalogValidator) and renders it as either a flat .csv
6
+ file (Table Validation data only) or a formatted, multi-sheet .xlsx
7
+ workbook (Summary / Table Validation / Column Validation / Data
8
+ Mismatches / Row Hash Mismatches).
9
+
10
+ Deliberately kept out of comparison_engine.py per the project spec:
11
+ the validator returns clean structured data, and this module is the
12
+ only place that knows about report formatting (csv / openpyxl).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import csv
18
+ import logging
19
+ from typing import Any, List, Optional, Tuple
20
+
21
+ from openpyxl import Workbook
22
+ from openpyxl.styles import Alignment, Border, Font, PatternFill, Side
23
+ from openpyxl.utils import get_column_letter
24
+ from openpyxl.worksheet.worksheet import Worksheet
25
+
26
+ from table_validator.models import CatalogValidationResponse, ValidationStatus
27
+
28
+ logger = logging.getLogger(__name__)
29
+
30
+ # ---------------------------------------------------------------------------
31
+ # Shared column definitions
32
+ # ---------------------------------------------------------------------------
33
+ TABLE_HEADERS = [
34
+ "Source Schema",
35
+ "Source Table",
36
+ "Target Schema",
37
+ "Target Table",
38
+ "Overall Status",
39
+ "Schema Match",
40
+ "Column Order",
41
+ "Row Count (Src)",
42
+ "Row Count (Tgt)",
43
+ "Row Count Diff",
44
+ "Data Types",
45
+ "Nullable",
46
+ "Null Counts",
47
+ "Distinct Counts",
48
+ "Min/Max",
49
+ "Data Match",
50
+ "Mismatch Count",
51
+ "Mismatch %",
52
+ "Row Hash Mismatch Count",
53
+ "Row Hash Mismatch %",
54
+ "Tier Reached",
55
+ "Partition",
56
+ "Validation Timestamp",
57
+ "Duration",
58
+ ]
59
+
60
+ # 1-based column indices within TABLE_HEADERS holding a PASS/FAIL/ERROR/SKIPPED value.
61
+ _TABLE_STATUS_COLUMNS = {5, 6, 7, 11, 12, 13, 14, 15, 16}
62
+
63
+ # Which ValidationType (config/schema.py) owns each 0-based TABLE_HEADERS
64
+ # column, for hiding columns whose validation type wasn't selected to run
65
+ # (see _filter_table_columns). None = always shown regardless of
66
+ # enabled_validations. "Schema Match"/"Column Order" measure column-name/
67
+ # order agreement (COLUMN, despite the "Schema Match" label being a
68
+ # pre-existing naming quirk from before source types other than
69
+ # Databricks existed) - not schema/table existence, which has no
70
+ # dedicated column here (it's surfaced as separate MISSING_FROM_TARGET/
71
+ # EXTRA_IN_TARGET rows instead, unaffected by column filtering).
72
+ _TABLE_COLUMN_OWNERS = [
73
+ None, None, None, None, None, # Source/Target Schema/Table, Overall Status
74
+ "column", # Schema Match
75
+ "column", # Column Order
76
+ "row", "row", "row", # Row Count (Src/Tgt/Diff)
77
+ "column", # Data Types
78
+ "column", # Nullable
79
+ "column", # Null Counts
80
+ "column", # Distinct Counts
81
+ "column", # Min/Max
82
+ "row", # Data Match
83
+ "row", "row", # Mismatch Count/%
84
+ "row", "row", # Row Hash Mismatch Count/%
85
+ "row", # Tier Reached
86
+ "row", # Partition
87
+ None, None, # Validation Timestamp, Duration
88
+ ]
89
+
90
+ COLUMN_HEADERS = [
91
+ "Source Schema", "Source Table", "Column", "Status",
92
+ "Source Type", "Target Type", "Type Status",
93
+ "Source Nullable", "Target Nullable", "Nullable Status",
94
+ "Source Nulls", "Target Nulls", "Null Status",
95
+ "Source Distinct", "Target Distinct", "Distinct Status",
96
+ "Source Min", "Source Max", "Target Min", "Target Max", "Min/Max Status",
97
+ "Error",
98
+ ]
99
+ _COLUMN_STATUS_COLUMNS = {4, 7, 10, 13, 16, 21}
100
+
101
+ MISMATCH_HEADERS = [
102
+ "Source Schema", "Source Table", "Primary Key",
103
+ "Mismatch Column", "Source Value", "Target Value",
104
+ "Row Hash (Source)", "Row Hash (Target)", "Verified",
105
+ ]
106
+
107
+ ROW_HASH_HEADERS = [
108
+ "Row #", "Source Schema", "Source Table",
109
+ "Primary Key", "Source Hash", "Target Hash", "Mismatch Status",
110
+ "Partition Bucket",
111
+ ]
112
+ _ROW_HASH_STATUS_COLUMN = 7
113
+
114
+ SUGGESTION_HEADERS = [
115
+ "Source Schema", "Source Table", "Column", "Issue Type", "Suggestion",
116
+ ]
117
+
118
+ SUMMARY_METRIC_LABELS = [
119
+ "Total Tables", "Passed Tables", "Failed Tables",
120
+ "Error Tables", "Skipped Tables", "Pass Percentage",
121
+ ]
122
+
123
+ # ---------------------------------------------------------------------------
124
+ # Styling
125
+ # ---------------------------------------------------------------------------
126
+ FONT_NAME = "Arial"
127
+ HEADER_FILL = PatternFill("solid", fgColor="1F4E78")
128
+ HEADER_FONT = Font(name=FONT_NAME, size=11, bold=True, color="FFFFFF")
129
+ TITLE_FONT = Font(name=FONT_NAME, size=16, bold=True, color="1F4E78")
130
+ LABEL_FONT = Font(name=FONT_NAME, size=10, bold=True)
131
+ VALUE_FONT = Font(name=FONT_NAME, size=10)
132
+ THIN_BORDER = Border(*(Side(style="thin", color="D9D9D9"),) * 4)
133
+
134
+ STATUS_FILLS = {
135
+ ValidationStatus.PASS: PatternFill("solid", fgColor="C6EFCE"),
136
+ ValidationStatus.FAIL: PatternFill("solid", fgColor="FFC7CE"),
137
+ ValidationStatus.ERROR: PatternFill("solid", fgColor="FFD8A8"),
138
+ ValidationStatus.SKIPPED: PatternFill("solid", fgColor="E7E6E6"),
139
+ }
140
+ STATUS_FONTS = {
141
+ ValidationStatus.PASS: Font(name=FONT_NAME, size=10, bold=True, color="006100"),
142
+ ValidationStatus.FAIL: Font(name=FONT_NAME, size=10, bold=True, color="9C0006"),
143
+ ValidationStatus.ERROR: Font(name=FONT_NAME, size=10, bold=True, color="974706"),
144
+ ValidationStatus.SKIPPED: Font(name=FONT_NAME, size=10, bold=True, color="666666"),
145
+ }
146
+
147
+ # Sort rank so FAIL/ERROR surface above PASS/SKIPPED within a schema.
148
+ _STATUS_SORT_RANK = {
149
+ "FAIL": 0,
150
+ "ERROR": 0,
151
+ "MISSING_FROM_TARGET": 0,
152
+ "EXTRA_IN_TARGET": 0,
153
+ "PASS": 1,
154
+ "SKIPPED": 1,
155
+ }
156
+
157
+ # Row-hash mismatch statuses aren't ValidationStatus members - map them onto
158
+ # the same PASS/FAIL/ERROR/SKIPPED fill+font conventions for the Row Hash
159
+ # Mismatches sheet (all three outcomes here are a real difference -> FAIL).
160
+ _ROW_HASH_STATUS_FILL_MAP = {
161
+ "MISMATCH": ValidationStatus.FAIL,
162
+ "MISSING_IN_TARGET": ValidationStatus.FAIL,
163
+ "MISSING_IN_SOURCE": ValidationStatus.FAIL,
164
+ "DUPLICATE_KEY": ValidationStatus.FAIL,
165
+ }
166
+
167
+
168
+ def _status_value(status: Optional[ValidationStatus]) -> str:
169
+ if status is None:
170
+ return ""
171
+ return status.value if isinstance(status, ValidationStatus) else str(status)
172
+
173
+
174
+ def _mismatch_count(table) -> Optional[int]:
175
+ if table.data is None:
176
+ return None
177
+ counts = [table.data.source_only_rows, table.data.target_only_rows, table.data.changed_rows]
178
+ if not all(c is None for c in counts):
179
+ return sum(c or 0 for c in counts)
180
+ # The tiered fail-fast funnel (Databricks-to-Databricks) never
181
+ # populates source_only_rows/target_only_rows/changed_rows - those
182
+ # only ever came from the legacy FULL-mode EXCEPT/hash-join path.
183
+ # Fall back to the row-hash comparison's own mismatch count, which is
184
+ # the real per-row mismatch figure for every table run through the
185
+ # tiered pipeline.
186
+ if table.data.row_hash_mismatch_count:
187
+ return table.data.row_hash_mismatch_count
188
+ return None
189
+
190
+
191
+ def _partition_summary(table) -> str:
192
+ if table.partitioned:
193
+ return (
194
+ f"{table.partition_column} "
195
+ f"({table.partition_buckets_culprit}/{table.partition_buckets_total} buckets differed)"
196
+ )
197
+ if table.partition_skip_reason:
198
+ return f"not partitioned ({table.partition_skip_reason})"
199
+ return ""
200
+
201
+
202
+ def _mismatch_pct(table, mismatch_count: Optional[int]) -> str:
203
+ if mismatch_count is None:
204
+ return ""
205
+ if table.data is not None and (
206
+ table.data.source_only_rows is None
207
+ and table.data.target_only_rows is None
208
+ and table.data.changed_rows is None
209
+ ):
210
+ # Tiered-pipeline fallback (see _mismatch_count): reuse the
211
+ # row-hash comparison's own percentage, which is computed over
212
+ # the union of keys actually compared on either side - more
213
+ # accurate than dividing by row_count_source alone when keys are
214
+ # missing/extra on one side.
215
+ return f"{table.data.row_hash_mismatch_percentage:.2f}%"
216
+ total = table.row_count_source or 0
217
+ if total <= 0:
218
+ return ""
219
+ return f"{(mismatch_count / total) * 100:.2f}%"
220
+
221
+
222
+ def _filter_table_columns(
223
+ headers: List[str],
224
+ rows: List[List[Any]],
225
+ enabled_validations: Optional[set] = None,
226
+ ) -> Tuple[List[str], List[List[Any]], set]:
227
+ """
228
+ Drop TABLE_HEADERS columns (and the matching cell in every row) whose
229
+ owning validation type isn't in enabled_validations - e.g. hide "Data
230
+ Types"/"Null Counts"/etc. entirely when "column" wasn't selected to
231
+ run, rather than showing a column of meaningless SKIPPED values.
232
+
233
+ enabled_validations=None means "no filtering" (show everything),
234
+ matching the CSV/legacy behavior and any caller that doesn't know
235
+ about per-run validation selection.
236
+
237
+ Returns (filtered_headers, filtered_rows, status_columns) where
238
+ status_columns is the 1-based set of surviving columns that hold a
239
+ PASS/FAIL/ERROR/SKIPPED value, recomputed for the new column
240
+ positions (_TABLE_STATUS_COLUMNS' original indices no longer apply
241
+ once columns are dropped).
242
+ """
243
+ if enabled_validations is None:
244
+ return headers, rows, _TABLE_STATUS_COLUMNS
245
+
246
+ keep_indices = [
247
+ i for i, owner in enumerate(_TABLE_COLUMN_OWNERS)
248
+ if owner is None or owner in enabled_validations
249
+ ]
250
+
251
+ filtered_headers = [headers[i] for i in keep_indices]
252
+ filtered_rows = [[row[i] for i in keep_indices] for row in rows]
253
+ status_columns = {
254
+ new_idx + 1
255
+ for new_idx, old_idx in enumerate(keep_indices)
256
+ if (old_idx + 1) in _TABLE_STATUS_COLUMNS
257
+ }
258
+
259
+ return filtered_headers, filtered_rows, status_columns
260
+
261
+
262
+ # ---------------------------------------------------------------------------
263
+ # Row builders (shared by CSV + every Excel sheet)
264
+ # ---------------------------------------------------------------------------
265
+ def _build_table_rows(result: CatalogValidationResponse) -> List[List[Any]]:
266
+ timestamp = result.validation_timestamp or ""
267
+ duration = result.execution_time_seconds
268
+
269
+ rows: List[List[Any]] = []
270
+ for schema in result.schemas:
271
+ for table in schema.tables:
272
+ data_status = table.data.status if table.data else None
273
+ mismatch_count = _mismatch_count(table)
274
+
275
+ row_hash_count = table.data.row_hash_mismatch_count if table.data else 0
276
+ row_hash_pct = (
277
+ f"{table.data.row_hash_mismatch_percentage:.2f}%" if table.data else ""
278
+ )
279
+
280
+ rows.append(
281
+ [
282
+ schema.schema_name,
283
+ table.table,
284
+ schema.schema_name,
285
+ table.table,
286
+ _status_value(table.status),
287
+ _status_value(table.columns_status),
288
+ _status_value(table.column_order_status),
289
+ table.row_count_source,
290
+ table.row_count_target,
291
+ table.row_count_difference,
292
+ _status_value(table.data_types_status),
293
+ _status_value(table.nullable_status),
294
+ _status_value(table.null_counts_status),
295
+ _status_value(table.distinct_counts_status),
296
+ _status_value(table.min_max_status),
297
+ _status_value(data_status),
298
+ mismatch_count if mismatch_count is not None else "",
299
+ _mismatch_pct(table, mismatch_count),
300
+ row_hash_count,
301
+ row_hash_pct,
302
+ table.tier_reached.name,
303
+ _partition_summary(table),
304
+ timestamp,
305
+ duration,
306
+ ]
307
+ )
308
+
309
+ for missing_table in schema.missing_tables:
310
+ rows.append(
311
+ [
312
+ schema.schema_name, missing_table, schema.schema_name, missing_table,
313
+ "MISSING_FROM_TARGET",
314
+ ]
315
+ + [""] * 17
316
+ + [timestamp, duration]
317
+ )
318
+
319
+ for extra_table in schema.extra_tables:
320
+ rows.append(
321
+ [
322
+ schema.schema_name, extra_table, schema.schema_name, extra_table,
323
+ "EXTRA_IN_TARGET",
324
+ ]
325
+ + [""] * 17
326
+ + [timestamp, duration]
327
+ )
328
+
329
+ # Sort: Source Schema asc, then FAIL/ERROR before PASS/SKIPPED, then Source Table asc.
330
+ rows.sort(key=lambda r: (r[0], _STATUS_SORT_RANK.get(r[4], 1), r[1]))
331
+ return rows
332
+
333
+
334
+ def _build_column_rows(result: CatalogValidationResponse) -> List[List[Any]]:
335
+ rows: List[List[Any]] = []
336
+ for schema in result.schemas:
337
+ for table in schema.tables:
338
+ for col in table.columns:
339
+ rows.append(
340
+ [
341
+ schema.schema_name, table.table, col.column, _status_value(col.status),
342
+ col.source_data_type, col.target_data_type, _status_value(col.data_type_status),
343
+ col.source_nullable, col.target_nullable, _status_value(col.nullable_status),
344
+ col.source_null_count, col.target_null_count, _status_value(col.null_count_status),
345
+ col.source_distinct_count, col.target_distinct_count, _status_value(col.distinct_count_status),
346
+ col.source_min, col.source_max, col.target_min, col.target_max,
347
+ _status_value(col.min_max_status),
348
+ col.error or "",
349
+ ]
350
+ )
351
+
352
+ for missing_col in table.missing_columns:
353
+ rows.append(
354
+ [schema.schema_name, table.table, missing_col, "MISSING_FROM_TARGET"]
355
+ + [""] * 18
356
+ )
357
+ for extra_col in table.extra_columns:
358
+ rows.append(
359
+ [schema.schema_name, table.table, extra_col, "EXTRA_IN_TARGET"]
360
+ + [""] * 18
361
+ )
362
+
363
+ rows.sort(key=lambda r: (r[0], r[1], r[2]))
364
+ return rows
365
+
366
+
367
+ def _build_mismatch_rows(result: CatalogValidationResponse) -> List[List[Any]]:
368
+ rows: List[List[Any]] = []
369
+ for schema in result.schemas:
370
+ for table in schema.tables:
371
+ if table.data is None:
372
+ continue
373
+ for detail in table.data.sample_changed_detail:
374
+ key_text = ", ".join(f"{k}={v}" for k, v in detail.primary_key.items())
375
+ rows.append(
376
+ [
377
+ schema.schema_name,
378
+ table.table,
379
+ key_text,
380
+ detail.mismatch_column,
381
+ detail.source_value,
382
+ detail.target_value,
383
+ detail.source_row_hash,
384
+ detail.target_row_hash,
385
+ "Yes" if detail.verified else "No (row-number, unverified)",
386
+ ]
387
+ )
388
+
389
+ # Group by Source Schema, Source Table, Primary Key.
390
+ rows.sort(key=lambda r: (r[0], r[1], r[2]))
391
+ return rows
392
+
393
+
394
+ def _build_row_hash_rows(result: CatalogValidationResponse) -> List[List[Any]]:
395
+ rows: List[List[Any]] = []
396
+ for schema in result.schemas:
397
+ for table in schema.tables:
398
+ if table.data is None:
399
+ continue
400
+ for mismatch in table.data.row_hash_mismatches:
401
+ rows.append(
402
+ [
403
+ schema.schema_name,
404
+ table.table,
405
+ mismatch.primary_key,
406
+ mismatch.source_hash,
407
+ mismatch.target_hash,
408
+ mismatch.status,
409
+ mismatch.partition_bucket or "",
410
+ ]
411
+ )
412
+
413
+ # Group by Source Schema, Source Table, Primary Key.
414
+ rows.sort(key=lambda r: (r[0], r[1], r[2]))
415
+
416
+ # Prepend a 1-based sequential row number, assigned after sorting so it
417
+ # reflects final display order (not discovery order).
418
+ return [[i, *row] for i, row in enumerate(rows, start=1)]
419
+
420
+
421
+ def _build_suggestion_rows(result: CatalogValidationResponse) -> List[List[Any]]:
422
+ """
423
+ One plain-English sentence per issue found on a table, covering every
424
+ category that can independently fail a table's Overall Status:
425
+ schema/constraint issues (missing/extra column, column order, data
426
+ type, nullable), per-column statistics (null count, distinct count,
427
+ min/max), row count, and row-level data (pointing at the Data
428
+ Mismatches / Row Hash Mismatches sheets for full detail rather than
429
+ duplicating it here).
430
+
431
+ If a table's row data actually matches (row count, row-hash/data
432
+ comparison all passed) but the table still failed purely on
433
+ schema/constraint or statistics grounds, an extra summary sentence
434
+ calls that out explicitly, since it's an easy thing to miss buried in
435
+ status columns. Every FAILed table gets at least one row here - if
436
+ none of the specific categories below apply, a fallback sentence
437
+ still points at Table Validation for the table's own Error field.
438
+ """
439
+ rows: List[List[Any]] = []
440
+
441
+ for schema in result.schemas:
442
+ for table in schema.tables:
443
+ if table.status not in (ValidationStatus.FAIL, ValidationStatus.ERROR):
444
+ continue
445
+
446
+ table_issues: List[List[Any]] = []
447
+
448
+ if table.schema_blocking:
449
+ # BLOCKING schema difference (Tier 0) - the table was
450
+ # aborted before any row-level tier ran, so there is no
451
+ # row-hash/data finding to report alongside this. Surface
452
+ # that explicitly rather than falling through to the
453
+ # generic per-category checks below, most of which won't
454
+ # have fired for a blocked table anyway.
455
+ table_issues.append([
456
+ schema.schema_name, table.table, "-", "Blocked at Schema Check",
457
+ f"Table '{table.table}' has a schema difference severe enough "
458
+ f"(missing/extra column, incompatible data type, or a missing "
459
+ f"configured key column) that row-level comparison was not "
460
+ f"attempted - resolve the schema issue(s) below first, then "
461
+ f"re-run to check the data.",
462
+ ])
463
+
464
+ for col in table.missing_columns:
465
+ table_issues.append([
466
+ schema.schema_name, table.table, col, "Missing Column",
467
+ f"Column '{col}' exists in the source but not in the target - "
468
+ f"it may not have been migrated, or was renamed/dropped.",
469
+ ])
470
+
471
+ for col in table.extra_columns:
472
+ table_issues.append([
473
+ schema.schema_name, table.table, col, "Extra Column",
474
+ f"Column '{col}' exists in the target but not in the source - "
475
+ f"check whether it was added intentionally or is leftover from a prior load.",
476
+ ])
477
+
478
+ if table.column_order_status == ValidationStatus.FAIL:
479
+ table_issues.append([
480
+ schema.schema_name, table.table, "-", "Column Order",
481
+ f"Columns match by name but are in a different order between source "
482
+ f"({', '.join(table.source_column_order)}) and target "
483
+ f"({', '.join(table.target_column_order)}).",
484
+ ])
485
+
486
+ if table.row_count_status == ValidationStatus.FAIL:
487
+ diff = table.row_count_difference
488
+ direction = "more" if (diff or 0) > 0 else "fewer"
489
+ table_issues.append([
490
+ schema.schema_name, table.table, "-", "Row Count Mismatch",
491
+ f"Target has {abs(diff) if diff is not None else 'a different number of'} "
492
+ f"{direction} rows than source ({table.row_count_source} vs "
493
+ f"{table.row_count_target}) - check for a partial load, duplicate rows, "
494
+ f"or rows deleted/inserted after migration.",
495
+ ])
496
+ elif table.row_count_status == ValidationStatus.ERROR:
497
+ table_issues.append([
498
+ schema.schema_name, table.table, "-", "Row Count Error",
499
+ f"Row count could not be verified for this table"
500
+ + (f": {table.error}" if table.error else " - see the Error column in Table Validation.")
501
+ ])
502
+
503
+ for col in table.columns:
504
+ if col.data_type_status == ValidationStatus.FAIL:
505
+ table_issues.append([
506
+ schema.schema_name, table.table, col.column, "Data Type Mismatch",
507
+ f"Column '{col.column}' is {col.source_data_type} in the source but "
508
+ f"{col.target_data_type} in the target - if the row data still matches, "
509
+ f"only the declared type differs; if not, check for precision/format loss "
510
+ f"during migration.",
511
+ ])
512
+
513
+ if col.nullable_status == ValidationStatus.FAIL:
514
+ src_null = "nullable" if col.source_nullable else "NOT NULL"
515
+ tgt_null = "nullable" if col.target_nullable else "NOT NULL"
516
+ table_issues.append([
517
+ schema.schema_name, table.table, col.column, "Nullable Mismatch",
518
+ f"Column '{col.column}' is declared {src_null} in the source but "
519
+ f"{tgt_null} in the target - this is a constraint difference, not "
520
+ f"necessarily a data problem, unless the stricter side is expected "
521
+ f"to reject values the other side allows.",
522
+ ])
523
+
524
+ if col.null_count_status == ValidationStatus.FAIL:
525
+ table_issues.append([
526
+ schema.schema_name, table.table, col.column, "Null Count Mismatch",
527
+ f"Column '{col.column}' has {col.source_null_count} NULLs in the source "
528
+ f"but {col.target_null_count} in the target - some rows likely gained or "
529
+ f"lost a NULL value for this column during migration.",
530
+ ])
531
+
532
+ if col.distinct_count_status == ValidationStatus.FAIL:
533
+ table_issues.append([
534
+ schema.schema_name, table.table, col.column, "Distinct Count Mismatch",
535
+ f"Column '{col.column}' has {col.source_distinct_count} distinct values "
536
+ f"in the source but {col.target_distinct_count} in the target - check for "
537
+ f"duplicate or collapsed values, or rows missing on one side.",
538
+ ])
539
+
540
+ if col.min_max_status == ValidationStatus.FAIL:
541
+ table_issues.append([
542
+ schema.schema_name, table.table, col.column, "Min/Max Mismatch",
543
+ f"Column '{col.column}' ranges from {col.source_min} to {col.source_max} "
544
+ f"in the source but {col.target_min} to {col.target_max} in the target - "
545
+ f"check for outlier rows unique to one side, or a truncated/extended value range.",
546
+ ])
547
+
548
+ data = table.data
549
+ if data is not None:
550
+ if data.row_hash_mismatch_count and data.row_hash_mismatch_count > 0:
551
+ by_row_number = data.key_columns == ["row_number"]
552
+ by_clause = (
553
+ "by row position (best-effort - no primary key "
554
+ "configured, so this cannot guarantee the same "
555
+ "record on both sides)"
556
+ if by_row_number
557
+ else "by primary key"
558
+ )
559
+ detail_clause = (
560
+ "values (marked unverified for row-position matches), if available"
561
+ if by_row_number
562
+ else "values, if available"
563
+ )
564
+ table_issues.append([
565
+ schema.schema_name, table.table, "-", "Row Data Mismatch",
566
+ f"{data.row_hash_mismatch_count} row(s) "
567
+ f"({data.row_hash_mismatch_percentage:.2f}%) differ between source and "
568
+ f"target {by_clause} - see the 'Row Hash Mismatches' sheet for the "
569
+ f"affected keys, and 'Data Mismatches' for the specific column(s) and "
570
+ f"{detail_clause}.",
571
+ ])
572
+ elif data.status == ValidationStatus.FAIL and (
573
+ (data.source_only_rows or 0) > 0 or (data.target_only_rows or 0) > 0
574
+ or (data.changed_rows or 0) > 0
575
+ ):
576
+ table_issues.append([
577
+ schema.schema_name, table.table, "-", "Row Data Mismatch",
578
+ f"Row-level comparison found {data.source_only_rows or 0} row(s) only in "
579
+ f"the source, {data.target_only_rows or 0} only in the target, and "
580
+ f"{data.changed_rows or 0} changed - see the 'Data Mismatches' sheet for detail.",
581
+ ])
582
+ elif data.status == ValidationStatus.ERROR:
583
+ table_issues.append([
584
+ schema.schema_name, table.table, "-", "Row Data Comparison Error",
585
+ f"Row-level comparison could not complete for this table"
586
+ + (f": {data.error}" if data.error else " - see the Error column in Table Validation.")
587
+ ])
588
+
589
+ if not table_issues:
590
+ # Table FAILed/ERRORed but none of the categories above
591
+ # explain why (e.g. an ERROR raised before any stage ran) -
592
+ # never leave a failed table unexplained.
593
+ table_issues.append([
594
+ schema.schema_name, table.table, "-", "Unclassified",
595
+ f"Table '{table.table}' has status {table.status.value} but the reason "
596
+ f"doesn't match a known category here - check the Error column and "
597
+ f"per-check status columns on the 'Table Validation' sheet for this table.",
598
+ ])
599
+
600
+ schema_only_issue = (
601
+ not table.schema_blocking
602
+ and table.status == ValidationStatus.FAIL
603
+ and table.row_count_status != ValidationStatus.FAIL
604
+ and (table.data is None or table.data.status != ValidationStatus.FAIL)
605
+ and not any(issue[3] == "Row Data Mismatch" for issue in table_issues)
606
+ )
607
+ if schema_only_issue:
608
+ rows.append([
609
+ schema.schema_name, table.table, "-", "Summary",
610
+ f"Table '{table.table}' failed, but the row data matches "
611
+ f"(row counts and row-hash/data comparison passed) - the failure "
612
+ f"is caused entirely by the schema/constraint issue(s) below.",
613
+ ])
614
+ rows.extend(table_issues)
615
+
616
+ rows.sort(key=lambda r: (r[0], r[1]))
617
+ return rows
618
+
619
+
620
+ def _build_summary_metrics(result: CatalogValidationResponse) -> List[Tuple[str, Any]]:
621
+ total = passed = failed = errors = skipped = 0
622
+ for schema in result.schemas:
623
+ for table in schema.tables:
624
+ total += 1
625
+ if table.status == ValidationStatus.PASS:
626
+ passed += 1
627
+ elif table.status == ValidationStatus.ERROR:
628
+ errors += 1
629
+ elif table.status == ValidationStatus.SKIPPED:
630
+ skipped += 1
631
+ else:
632
+ failed += 1
633
+ total += len(schema.missing_tables)
634
+ failed += len(schema.missing_tables)
635
+
636
+ pass_pct = f"{(passed / total) * 100:.2f}%" if total else "0.00%"
637
+
638
+ return [
639
+ ("Total Tables", total),
640
+ ("Passed Tables", passed),
641
+ ("Failed Tables", failed),
642
+ ("Error Tables", errors),
643
+ ("Skipped Tables", skipped),
644
+ ("Pass Percentage", pass_pct),
645
+ ]
646
+
647
+
648
+ # ---------------------------------------------------------------------------
649
+ # CSV (Table Validation data only)
650
+ # ---------------------------------------------------------------------------
651
+ def generate_csv_report(
652
+ result: CatalogValidationResponse,
653
+ output_path: str,
654
+ ) -> str:
655
+ """
656
+ Render the per-table validation results of a CatalogValidationResponse
657
+ as a flat .csv file. Returns the output_path for convenience.
658
+ """
659
+ logger.info(
660
+ "Generating CSV report | source=%s | target=%s | -> %s",
661
+ result.source_catalog, result.target_catalog, output_path,
662
+ )
663
+
664
+ with open(output_path, "w", newline="", encoding="utf-8-sig") as f:
665
+ writer = csv.writer(f)
666
+ writer.writerow(TABLE_HEADERS)
667
+ writer.writerows(_build_table_rows(result))
668
+
669
+ logger.info("CSV report written to %s", output_path)
670
+ return output_path
671
+
672
+
673
+ # ---------------------------------------------------------------------------
674
+ # Excel helpers
675
+ # ---------------------------------------------------------------------------
676
+ def _write_header_row(ws: Worksheet, headers: List[str]) -> None:
677
+ for col_idx, header in enumerate(headers, start=1):
678
+ cell = ws.cell(row=1, column=col_idx, value=header)
679
+ cell.font = HEADER_FONT
680
+ cell.fill = HEADER_FILL
681
+ cell.alignment = Alignment(horizontal="center", vertical="center", wrap_text=True)
682
+ cell.border = THIN_BORDER
683
+ ws.freeze_panes = ws.cell(row=2, column=1)
684
+
685
+
686
+ def _write_rows(
687
+ ws: Worksheet,
688
+ rows: List[List[Any]],
689
+ status_columns: set,
690
+ group_col: Optional[int] = None,
691
+ custom_status_map: Optional[dict] = None,
692
+ ) -> int:
693
+ """
694
+ Write rows starting at row 2. If group_col is set, consecutive rows
695
+ sharing that column's value are put into a collapsible outline group
696
+ (Excel row grouping) - one level, matching the schema boundary.
697
+ Returns the last row index written (1 if no data rows).
698
+ """
699
+ row_idx = 2
700
+ prev_group_value = None
701
+ group_start = None
702
+
703
+ def _close_group(end_row: int) -> None:
704
+ if group_start is not None and end_row > group_start:
705
+ for r in range(group_start, end_row):
706
+ ws.row_dimensions[r].outline_level = 1
707
+
708
+ for values in rows:
709
+ if group_col is not None:
710
+ current_value = values[group_col - 1]
711
+ if current_value != prev_group_value:
712
+ if prev_group_value is not None:
713
+ _close_group(row_idx)
714
+ prev_group_value = current_value
715
+ group_start = row_idx + 1 # first detail row after the group's own header row
716
+
717
+ for col_idx, value in enumerate(values, start=1):
718
+ cell = ws.cell(row=row_idx, column=col_idx, value=value)
719
+ cell.font = VALUE_FONT
720
+ cell.border = THIN_BORDER
721
+ cell.alignment = Alignment(vertical="center", wrap_text=False)
722
+
723
+ if col_idx in status_columns:
724
+ status = next((s for s in ValidationStatus if s.value == value), None)
725
+ if status is None and custom_status_map is not None:
726
+ status = custom_status_map.get(value)
727
+ if status in STATUS_FILLS:
728
+ cell.fill = STATUS_FILLS[status]
729
+ cell.font = STATUS_FONTS[status]
730
+ elif value in ("MISSING_FROM_TARGET", "EXTRA_IN_TARGET"):
731
+ cell.fill = STATUS_FILLS[ValidationStatus.FAIL]
732
+ cell.font = STATUS_FONTS[ValidationStatus.FAIL]
733
+
734
+ row_idx += 1
735
+
736
+ if group_col is not None:
737
+ _close_group(row_idx)
738
+
739
+ return row_idx - 1
740
+
741
+
742
+ def _autofit(ws: Worksheet, headers: List[str], rows: List[List[Any]]) -> None:
743
+ for col_idx, header in enumerate(headers, start=1):
744
+ max_len = len(str(header))
745
+ for row_values in rows:
746
+ value = row_values[col_idx - 1]
747
+ if value is not None:
748
+ max_len = max(max_len, len(str(value)))
749
+ ws.column_dimensions[get_column_letter(col_idx)].width = min(max_len + 2, 60)
750
+
751
+
752
+ def _enable_filter(ws: Worksheet, num_cols: int, last_row: int) -> None:
753
+ ws.auto_filter.ref = f"A1:{get_column_letter(num_cols)}{max(last_row, 1)}"
754
+
755
+
756
+ # ---------------------------------------------------------------------------
757
+ # Excel sheet builders
758
+ # ---------------------------------------------------------------------------
759
+ def _build_summary_sheet(
760
+ wb: Workbook,
761
+ result: CatalogValidationResponse,
762
+ source_type: Optional[str] = None,
763
+ validations_run: Optional[str] = None,
764
+ ) -> None:
765
+ ws = wb.active
766
+ ws.title = "Summary"
767
+
768
+ ws["A1"] = "Databricks Catalog Validation Report"
769
+ ws["A1"].font = TITLE_FONT
770
+ ws.merge_cells("A1:B1")
771
+
772
+ ws["A2"] = f"Source: {result.source_catalog} -> Target: {result.target_catalog}"
773
+ ws["A2"].font = VALUE_FONT
774
+ ws.merge_cells("A2:B2")
775
+
776
+ row = 4
777
+ fields = [("Overall Status", _status_value(result.status))]
778
+ if source_type:
779
+ fields.append(("Source Type", source_type))
780
+ if validations_run:
781
+ fields.append(("Validations Run", validations_run))
782
+ fields += [
783
+ ("Validation Timestamp", result.validation_timestamp or ""),
784
+ ("Duration (s)", result.execution_time_seconds),
785
+ ]
786
+ for label, value in fields:
787
+ ws.cell(row=row, column=1, value=label).font = LABEL_FONT
788
+ cell = ws.cell(row=row, column=2, value=value)
789
+ cell.font = VALUE_FONT
790
+ if label == "Overall Status":
791
+ status = next((s for s in ValidationStatus if s.value == value), None)
792
+ if status in STATUS_FILLS:
793
+ cell.fill = STATUS_FILLS[status]
794
+ cell.font = STATUS_FONTS[status]
795
+ row += 1
796
+
797
+ row += 1
798
+ ws.cell(row=row, column=1, value="Table Summary").font = Font(
799
+ name=FONT_NAME, size=12, bold=True, color="1F4E78"
800
+ )
801
+ row += 1
802
+
803
+ header_row = row
804
+ ws.cell(row=header_row, column=1, value="Metric").font = HEADER_FONT
805
+ ws.cell(row=header_row, column=1).fill = HEADER_FILL
806
+ ws.cell(row=header_row, column=2, value="Value").font = HEADER_FONT
807
+ ws.cell(row=header_row, column=2).fill = HEADER_FILL
808
+ row += 1
809
+
810
+ for label, value in _build_summary_metrics(result):
811
+ ws.cell(row=row, column=1, value=label).font = VALUE_FONT
812
+ cell = ws.cell(row=row, column=2, value=value)
813
+ cell.font = VALUE_FONT
814
+ cell.alignment = Alignment(horizontal="right")
815
+ row += 1
816
+
817
+ ws.column_dimensions["A"].width = 28
818
+ ws.column_dimensions["B"].width = 40
819
+
820
+
821
+ def _build_table_validation_sheet(
822
+ wb: Workbook,
823
+ result: CatalogValidationResponse,
824
+ enabled_validations: Optional[set] = None,
825
+ ) -> None:
826
+ ws = wb.create_sheet("Table Validation")
827
+ rows = _build_table_rows(result)
828
+ headers, rows, status_columns = _filter_table_columns(
829
+ TABLE_HEADERS, rows, enabled_validations
830
+ )
831
+
832
+ _write_header_row(ws, headers)
833
+ last_row = _write_rows(ws, rows, status_columns, group_col=1)
834
+ _autofit(ws, headers, rows)
835
+ _enable_filter(ws, len(headers), last_row)
836
+
837
+
838
+ def _build_column_validation_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
839
+ ws = wb.create_sheet("Column Validation")
840
+ rows = _build_column_rows(result)
841
+
842
+ _write_header_row(ws, COLUMN_HEADERS)
843
+ last_row = _write_rows(ws, rows, _COLUMN_STATUS_COLUMNS, group_col=1)
844
+ _autofit(ws, COLUMN_HEADERS, rows)
845
+ _enable_filter(ws, len(COLUMN_HEADERS), last_row)
846
+
847
+
848
+ def _build_data_mismatches_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
849
+ ws = wb.create_sheet("Data Mismatches")
850
+ rows = _build_mismatch_rows(result)
851
+
852
+ _write_header_row(ws, MISMATCH_HEADERS)
853
+ last_row = _write_rows(ws, rows, set(), group_col=1)
854
+ _autofit(ws, MISMATCH_HEADERS, rows)
855
+ _enable_filter(ws, len(MISMATCH_HEADERS), last_row)
856
+
857
+
858
+ def _build_row_hash_mismatches_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
859
+ ws = wb.create_sheet("Row Hash Mismatches")
860
+ rows = _build_row_hash_rows(result)
861
+
862
+ _write_header_row(ws, ROW_HASH_HEADERS)
863
+ last_row = _write_rows(
864
+ ws, rows, {_ROW_HASH_STATUS_COLUMN}, group_col=2,
865
+ custom_status_map=_ROW_HASH_STATUS_FILL_MAP,
866
+ )
867
+ _autofit(ws, ROW_HASH_HEADERS, rows)
868
+ _enable_filter(ws, len(ROW_HASH_HEADERS), last_row)
869
+
870
+
871
+ def _build_suggestions_sheet(wb: Workbook, result: CatalogValidationResponse) -> None:
872
+ ws = wb.create_sheet("Suggestions")
873
+ rows = _build_suggestion_rows(result)
874
+
875
+ _write_header_row(ws, SUGGESTION_HEADERS)
876
+
877
+ row_idx = 2
878
+ for values in rows:
879
+ for col_idx, value in enumerate(values, start=1):
880
+ cell = ws.cell(row=row_idx, column=col_idx, value=value)
881
+ cell.font = VALUE_FONT
882
+ cell.border = THIN_BORDER
883
+ cell.alignment = Alignment(vertical="top", wrap_text=(col_idx == len(SUGGESTION_HEADERS)))
884
+ row_idx += 1
885
+
886
+ ws.column_dimensions[get_column_letter(1)].width = 18
887
+ ws.column_dimensions[get_column_letter(2)].width = 20
888
+ ws.column_dimensions[get_column_letter(3)].width = 16
889
+ ws.column_dimensions[get_column_letter(4)].width = 18
890
+ ws.column_dimensions[get_column_letter(5)].width = 90
891
+
892
+ ws.freeze_panes = ws.cell(row=2, column=1)
893
+ ws.auto_filter.ref = f"A1:{get_column_letter(len(SUGGESTION_HEADERS))}{max(row_idx - 1, 1)}"
894
+
895
+
896
+ # ---------------------------------------------------------------------------
897
+ # Public entry point
898
+ # ---------------------------------------------------------------------------
899
+ def generate_excel_report(
900
+ result: CatalogValidationResponse,
901
+ output_path: str,
902
+ source_type: Optional[str] = None,
903
+ enabled_validations: Optional[set] = None,
904
+ ) -> str:
905
+ """
906
+ Render a CatalogValidationResponse as a formatted, multi-sheet .xlsx
907
+ workbook: Summary, Table Validation, Column Validation, Data
908
+ Mismatches, Row Hash Mismatches, Suggestions. Returns the output_path
909
+ for convenience.
910
+
911
+ source_type (e.g. "databricks"/"azure_blob"/"azure_sql") is optional
912
+ and purely cosmetic - shown on the Summary sheet next to Overall
913
+ Status so it's clear at a glance what kind of source was compared,
914
+ since the report format itself is identical regardless of source.
915
+
916
+ enabled_validations (e.g. {"catalog", "row"}) is optional; when given,
917
+ Table Validation columns and whole sheets belonging to a validation
918
+ type NOT in this set are omitted entirely - "column" gates the
919
+ Column Validation sheet plus the column-level columns on Table
920
+ Validation, "row" gates Data Mismatches/Row Hash Mismatches plus the
921
+ row-level columns on Table Validation. None means "show everything"
922
+ (no filtering), matching prior behavior for any caller that doesn't
923
+ pass it.
924
+ """
925
+ logger.info(
926
+ "Generating Excel report | source=%s | target=%s | source_type=%s | "
927
+ "enabled_validations=%s | -> %s",
928
+ result.source_catalog, result.target_catalog, source_type,
929
+ enabled_validations, output_path,
930
+ )
931
+
932
+ validations_run = (
933
+ ", ".join(sorted(enabled_validations)) if enabled_validations is not None else None
934
+ )
935
+
936
+ wb = Workbook()
937
+
938
+ _build_summary_sheet(wb, result, source_type, validations_run)
939
+ _build_table_validation_sheet(wb, result, enabled_validations)
940
+
941
+ if enabled_validations is None or "column" in enabled_validations:
942
+ _build_column_validation_sheet(wb, result)
943
+
944
+ if enabled_validations is None or "row" in enabled_validations:
945
+ _build_data_mismatches_sheet(wb, result)
946
+ _build_row_hash_mismatches_sheet(wb, result)
947
+
948
+ _build_suggestions_sheet(wb, result)
949
+
950
+ wb.save(output_path)
951
+
952
+ logger.info("Excel report written to %s", output_path)
953
+ return output_path