table-validator 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,952 @@
1
+ """
2
+ Pydantic models for the Data Migration Comparison Service.
3
+
4
+ Defines request / response contracts used by the FastAPI layer and the
5
+ comparison engine.
6
+
7
+ This file contains two families of models:
8
+
9
+ 1. Existing CSV-vs-Databricks row comparison models (CompareRequest /
10
+ ComparisonResult / etc.) - UNCHANGED from the original implementation.
11
+
12
+ 2. New Databricks catalog-to-catalog validation models, added to support
13
+ CatalogValidator in comparison_engine.py. These are intentionally kept
14
+ separate (different enum, different result shape) rather than
15
+ shoehorned into the existing ComparisonStatus / ComparisonResult
16
+ models, since a catalog validation run produces a tree of results
17
+ (catalog -> schemas -> tables -> columns) rather than a single flat
18
+ comparison.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from enum import Enum
24
+ from typing import Any, Dict, List, Optional, Set
25
+
26
+ from pydantic import BaseModel, Field, field_validator
27
+
28
+ from table_validator.config.schema import ValidationType
29
+
30
+
31
+ # ---------------------------------------------------------------------------
32
+ # Enumerations
33
+ # ---------------------------------------------------------------------------
34
+ class Platform(str, Enum):
35
+ AZURE_STORAGE = "azure_storage"
36
+ DATABRICKS = "databricks"
37
+
38
+
39
+ class ComparisonStatus(str, Enum):
40
+ PASS = "PASS"
41
+ WARN = "WARN"
42
+ FAIL = "FAIL"
43
+
44
+
45
+ # ---------------------------------------------------------------------------
46
+ # Nested configuration models
47
+ # ---------------------------------------------------------------------------
48
+ class SourceConfig(BaseModel):
49
+
50
+ platform: Platform = Field(
51
+ default=Platform.AZURE_STORAGE,
52
+ description="Source platform identifier",
53
+ )
54
+
55
+ table: str = Field(
56
+ ...,
57
+ min_length=1,
58
+ description="Source CSV file path inside Azure Storage",
59
+ )
60
+
61
+ query: Optional[str] = Field(
62
+ default=None,
63
+ description="Reserved for future use.",
64
+ )
65
+
66
+ model_config = {"extra": "forbid"}
67
+
68
+
69
+ class TargetConfig(BaseModel):
70
+
71
+ platform: Platform = Field(
72
+ default=Platform.DATABRICKS,
73
+ description="Target platform identifier",
74
+ )
75
+
76
+ table: str = Field(
77
+ ...,
78
+ min_length=1,
79
+ description="Target Databricks table name",
80
+ )
81
+
82
+ query: Optional[str] = Field(
83
+ default=None,
84
+ description="Optional Databricks SQL query.",
85
+ )
86
+
87
+ model_config = {"extra": "forbid"}
88
+
89
+
90
+ class ComparisonOptions(BaseModel):
91
+
92
+ primary_keys: List[str] = Field(
93
+ default_factory=list,
94
+ description="Primary key column(s).",
95
+ )
96
+
97
+ ignore_columns: List[str] = Field(
98
+ default_factory=list,
99
+ description="Columns ignored during comparison.",
100
+ )
101
+
102
+ compare_schema: bool = Field(
103
+ default=True,
104
+ description="Enable schema comparison",
105
+ )
106
+
107
+ compare_values: bool = Field(
108
+ default=True,
109
+ description="Enable value comparison",
110
+ )
111
+
112
+ compare_duplicates: bool = Field(
113
+ default=True,
114
+ description="Enable duplicate detection",
115
+ )
116
+
117
+ case_sensitive: bool = Field(
118
+ default=True,
119
+ description="Case-sensitive comparison",
120
+ )
121
+
122
+ trim_strings: bool = Field(
123
+ default=True,
124
+ description="Trim leading/trailing spaces before comparison",
125
+ )
126
+
127
+ numeric_tolerance: float = Field(
128
+ default=0.0,
129
+ ge=0.0,
130
+ description="Numeric comparison tolerance",
131
+ )
132
+
133
+ sample_size: int = Field(
134
+ default=20,
135
+ ge=1,
136
+ le=500,
137
+ description="Maximum mismatch samples returned",
138
+ )
139
+
140
+ model_config = {"extra": "forbid"}
141
+
142
+
143
+ # ---------------------------------------------------------------------------
144
+ # Request model
145
+ # ---------------------------------------------------------------------------
146
+ class CompareRequest(BaseModel):
147
+
148
+ source_platform: Platform = Field(
149
+ default=Platform.AZURE_STORAGE,
150
+ description="Source platform",
151
+ )
152
+
153
+ target_platform: Platform = Field(
154
+ default=Platform.DATABRICKS,
155
+ description="Target platform",
156
+ )
157
+
158
+ source_table: str = Field(
159
+ ...,
160
+ min_length=1,
161
+ description="Azure Storage CSV path (example: n8ndirectory/day.csv)",
162
+ )
163
+
164
+ target_table: str = Field(
165
+ ...,
166
+ min_length=1,
167
+ description="Databricks table name",
168
+ )
169
+
170
+ source_query: Optional[str] = Field(
171
+ default=None,
172
+ description="Reserved for future use.",
173
+ )
174
+
175
+ target_query: Optional[str] = Field(
176
+ default=None,
177
+ description="Optional Databricks SQL query.",
178
+ )
179
+
180
+ primary_keys: List[str] = Field(
181
+ default_factory=list,
182
+ description="Primary key column(s)",
183
+ )
184
+
185
+ ignore_columns: List[str] = Field(
186
+ default_factory=list,
187
+ description="Columns ignored during comparison",
188
+ )
189
+
190
+ compare_schema: bool = Field(default=True)
191
+ compare_values: bool = Field(default=True)
192
+ compare_duplicates: bool = Field(default=True)
193
+ case_sensitive: bool = Field(default=True)
194
+
195
+ trim_strings: bool = Field(
196
+ default=True,
197
+ description="Trim whitespace before comparison",
198
+ )
199
+
200
+ numeric_tolerance: float = Field(
201
+ default=0.0,
202
+ ge=0.0,
203
+ )
204
+
205
+ sample_size: int = Field(
206
+ default=20,
207
+ ge=1,
208
+ le=500,
209
+ )
210
+
211
+ @classmethod
212
+ def from_configs(
213
+ cls,
214
+ source: SourceConfig,
215
+ target: TargetConfig,
216
+ options: Optional[ComparisonOptions] = None,
217
+ ) -> "CompareRequest":
218
+ opts = options or ComparisonOptions()
219
+ return cls(
220
+ source_platform=source.platform,
221
+ target_platform=target.platform,
222
+ source_table=source.table,
223
+ target_table=target.table,
224
+ source_query=source.query,
225
+ target_query=target.query,
226
+ primary_keys=opts.primary_keys,
227
+ ignore_columns=opts.ignore_columns,
228
+ compare_schema=opts.compare_schema,
229
+ compare_values=opts.compare_values,
230
+ compare_duplicates=opts.compare_duplicates,
231
+ case_sensitive=opts.case_sensitive,
232
+ trim_strings=opts.trim_strings,
233
+ numeric_tolerance=opts.numeric_tolerance,
234
+ sample_size=opts.sample_size,
235
+ )
236
+
237
+ @field_validator("primary_keys", "ignore_columns", mode="before")
238
+ @classmethod
239
+ def _ensure_list(cls, value: Any) -> List[str]:
240
+ if value is None:
241
+ return []
242
+ if isinstance(value, str):
243
+ return [value]
244
+ return list(value)
245
+
246
+ model_config = {
247
+ "extra": "forbid",
248
+ "json_schema_extra": {
249
+ "example": {
250
+ "source_platform": "azure_storage",
251
+ "target_platform": "databricks",
252
+ "source_table": "n8ndirectory/day.csv",
253
+ "target_table": "for_n8n_catalog.for_n8n_scheme.day",
254
+ "primary_keys": ["Numeric"],
255
+ "ignore_columns": [],
256
+ "compare_schema": True,
257
+ "compare_values": True,
258
+ "compare_duplicates": True,
259
+ "case_sensitive": False,
260
+ "trim_strings": True,
261
+ "numeric_tolerance": 0.01,
262
+ "sample_size": 20,
263
+ }
264
+ },
265
+ }
266
+
267
+
268
+ # ---------------------------------------------------------------------------
269
+ # Response models
270
+ # ---------------------------------------------------------------------------
271
+ class ComparisonResult(BaseModel):
272
+
273
+ status: ComparisonStatus = Field(
274
+ ...,
275
+ description="Overall comparison outcome",
276
+ )
277
+
278
+ execution_time: float = Field(
279
+ ...,
280
+ alias="execution_time_seconds",
281
+ ge=0.0,
282
+ )
283
+
284
+ row_count_source: int = Field(..., ge=0)
285
+ row_count_target: int = Field(..., ge=0)
286
+
287
+ matched_rows: int = Field(..., ge=0)
288
+ missing_rows: int = Field(..., ge=0)
289
+ extra_rows: int = Field(..., ge=0)
290
+
291
+ duplicate_rows: Dict[str, Any] = Field(
292
+ default_factory=dict,
293
+ )
294
+
295
+ schema_match: bool
296
+
297
+ column_differences: List[Dict[str, Any]] = Field(
298
+ default_factory=list,
299
+ )
300
+
301
+ sample_mismatches: List[Dict[str, Any]] = Field(
302
+ default_factory=list,
303
+ )
304
+
305
+ model_config = {
306
+ "populate_by_name": True,
307
+ "extra": "ignore",
308
+ }
309
+
310
+
311
+ class HealthResponse(BaseModel):
312
+
313
+ status: str = Field(
314
+ default="healthy",
315
+ examples=["healthy"],
316
+ )
317
+
318
+ model_config = {"extra": "forbid"}
319
+
320
+
321
+ # ---------------------------------------------------------------------------
322
+ # Backward compatibility aliases
323
+ # ---------------------------------------------------------------------------
324
+ ComparisonRequest = CompareRequest
325
+ ComparisonResponse = ComparisonResult
326
+
327
+
328
+ # ===========================================================================
329
+ # NEW: Databricks catalog-to-catalog validation models
330
+ # ===========================================================================
331
+ class ValidationStatus(str, Enum):
332
+ """
333
+ Status for a single validation stage or an aggregated object
334
+ (column / table / schema / catalog).
335
+
336
+ Distinct from ComparisonStatus (PASS/WARN/FAIL) because the catalog
337
+ validator needs to separate genuine validation failures from
338
+ technical errors (permission denied, connection dropped, etc.) and
339
+ from stages that were intentionally skipped by configuration.
340
+ """
341
+
342
+ PASS = "PASS"
343
+ FAIL = "FAIL"
344
+ ERROR = "ERROR"
345
+ SKIPPED = "SKIPPED"
346
+
347
+
348
+ class DataCompareMode(str, Enum):
349
+ """
350
+ Controls how expensive stage 15 (actual row-level data comparison) is.
351
+ Everything runs as push-down SQL against Databricks; no full-table
352
+ collect() / toPandas() is ever performed regardless of mode.
353
+ """
354
+
355
+ COUNT_ONLY = "COUNT_ONLY" # row count only, skip null/distinct/minmax/data
356
+ STATISTICS = "STATISTICS" # null/distinct/minmax, skip row-level data compare
357
+ HASH = "HASH" # key + row-hash based row compare (pushed down)
358
+ FULL = "FULL" # key-based anti-join, returns sample mismatched rows
359
+
360
+ @classmethod
361
+ def default(cls) -> "DataCompareMode":
362
+ # Safe default for large datasets: aggregate statistics, no
363
+ # row-level comparison.
364
+ return cls.STATISTICS
365
+
366
+
367
+ class ValidationTier(int, Enum):
368
+ """
369
+ How far the tiered fail-fast funnel got for one table (Databricks ->
370
+ Databricks path only, see CatalogValidator). Each tier only runs if
371
+ every cheaper tier before it failed to conclusively answer "are these
372
+ tables equal" - a table's tier_reached tells a reader exactly how much
373
+ work was (and, just as importantly, was NOT) done to reach its
374
+ verdict, which the previous "everything always runs" pipeline had no
375
+ way to express.
376
+
377
+ Tier 3 (partition/bucket fingerprinting) is deferred to a follow-up;
378
+ a mismatch at Tier 2 goes straight to Tier 4 over the whole table.
379
+ """
380
+
381
+ SCHEMA_BLOCKED = 0 # Tier 0 found a BLOCKING schema diff; aborted, no further tier ran
382
+ SCHEMA_ONLY = 1 # Tier 0 was the final word (e.g. ROW validation disabled)
383
+ STATISTICAL = 2 # Tier 1 was the final word (mismatch, or max_tier capped here)
384
+ FINGERPRINT = 3 # Tier 2 was the final word (whole-table fingerprint matched)
385
+ ROW_HASH = 4 # Tier 4 ran (per-key row-hash diff)
386
+ COLUMN_DIFF = 5 # Tier 5 ran (column-level diff of mismatched keys)
387
+
388
+
389
+ class HashCanonicalizationSpec(BaseModel):
390
+ """
391
+ Canonicalization rules shared by every hashing tier (Tier 2 whole-table
392
+ fingerprint, Tier 4 per-key row hash) so they can never disagree with
393
+ each other about what "the same row" hashes to.
394
+
395
+ Defaults reproduce the pre-existing, empirically-verified hash
396
+ expression byte-for-byte (see databricks_connector.get_row_hashes'
397
+ history and table_validator/CLAUDE.md's note on sha2()/hashlib.sha256
398
+ cross-dialect equivalence) - nothing changes until a caller
399
+ deliberately constructs a non-default spec. Not yet exposed via CLI/
400
+ wizard: these knobs are correctness-sensitive and need dedicated
401
+ verification against real data before users can toggle them.
402
+ """
403
+
404
+ null_sentinel: str = "\x01NULL\x01"
405
+ trim_strings: bool = False
406
+ case_sensitive: bool = True
407
+ float_rounding_decimals: Optional[int] = None
408
+ normalize_negative_zero: bool = False
409
+ unicode_nfc_normalize: bool = False
410
+
411
+ model_config = {"extra": "forbid"}
412
+
413
+
414
+ class PartitionPromptContext(BaseModel):
415
+ """
416
+ Passed to CatalogValidator's optional partition_prompt callback when a
417
+ table is both large (row count over request.partition_threshold) and
418
+ has a confirmed mismatch (Tier 1 and/or Tier 2) - i.e. exactly the
419
+ case where an unpartitioned Tier 4 row-hash diff would be expensive
420
+ and a bucketed comparison is worth offering. The callback returns the
421
+ chosen partition column name, or None to decline (falls back to
422
+ today's unpartitioned Tier 4 over the whole table).
423
+
424
+ Kept as a plain data-carrier with no behavior, so the validator (pure
425
+ decision logic, no I/O) and the CLI (the only place that actually
426
+ prompts a human) stay cleanly separated - the validator only ever
427
+ calls this callback and reads its return value.
428
+ """
429
+
430
+ schema_name: str
431
+ table: str
432
+ row_count: int
433
+ candidate_columns: List[str]
434
+
435
+ model_config = {"extra": "forbid"}
436
+
437
+
438
+ # ---------------------------------------------------------------------------
439
+ # Request
440
+ # ---------------------------------------------------------------------------
441
+ class CatalogValidationRequest(BaseModel):
442
+
443
+ source_catalog: str = Field(..., min_length=1)
444
+ target_catalog: str = Field(..., min_length=1)
445
+
446
+ schemas: Optional[List[str]] = Field(
447
+ default=None,
448
+ description=(
449
+ "Restrict validation to these schemas. If omitted, all "
450
+ "schemas common to both catalogs are validated."
451
+ ),
452
+ )
453
+
454
+ tables: Optional[List[str]] = Field(
455
+ default=None,
456
+ description=(
457
+ "Restrict validation to these table names (applies within "
458
+ "every validated schema). If omitted, all common tables are "
459
+ "validated."
460
+ ),
461
+ )
462
+
463
+ ignore_columns: List[str] = Field(default_factory=list)
464
+
465
+ case_sensitive_columns: bool = Field(
466
+ default=False,
467
+ description="Case-sensitive column name comparison.",
468
+ )
469
+
470
+ validate_column_order: bool = Field(
471
+ default=True,
472
+ description="If False, column order differences do not fail a table.",
473
+ )
474
+
475
+ validate_nullable: bool = Field(default=True)
476
+
477
+ primary_keys: Dict[str, List[str]] = Field(
478
+ default_factory=dict,
479
+ description=(
480
+ "Optional map of 'schema.table' -> key column list, used for "
481
+ "row-level data comparison (HASH / FULL modes). Tables without "
482
+ "an entry fall back to a safe COUNT_ONLY-style comparison for "
483
+ "stage 15."
484
+ ),
485
+ )
486
+
487
+ data_compare_mode: DataCompareMode = Field(
488
+ default_factory=DataCompareMode.default
489
+ )
490
+
491
+ max_tier: ValidationTier = Field(
492
+ default=ValidationTier.COLUMN_DIFF,
493
+ description=(
494
+ "Ceiling for the tiered fail-fast funnel (Tier 0 schema -> "
495
+ "Tier 1 statistics -> Tier 2 fingerprint -> Tier 4 row-hash -> "
496
+ "Tier 5 column diff). STATISTICAL stops after Tier 1 even if "
497
+ "it matches (the --mode=stats CLI flag); COLUMN_DIFF (default) "
498
+ "lets the funnel run all the way to a column-level diff if "
499
+ "cheaper tiers can't already prove the tables equal. Only "
500
+ "takes effect when ROW validation is enabled - it is inert "
501
+ "otherwise, same as data_compare_mode was."
502
+ ),
503
+ )
504
+
505
+ max_sample_rows: int = Field(
506
+ default=50,
507
+ ge=1,
508
+ le=1000,
509
+ description="Max sample mismatched rows returned per table in FULL mode.",
510
+ )
511
+
512
+ partition_threshold: int = Field(
513
+ default=1_000_000,
514
+ ge=1,
515
+ description=(
516
+ "Row-count threshold above which a table with a confirmed "
517
+ "mismatch (Tier 1 and/or Tier 2) is offered for partitioned "
518
+ "Tier 4 row-hash comparison via the partition_prompt callback, "
519
+ "instead of an unpartitioned whole-table scan. Below this "
520
+ "threshold, or when partition_prompt is not supplied/declines, "
521
+ "Tier 4 always runs unpartitioned as before."
522
+ ),
523
+ )
524
+
525
+ enabled_validations: Set[ValidationType] = Field(
526
+ default_factory=lambda: {
527
+ ValidationType.CATALOG,
528
+ ValidationType.SCHEMA,
529
+ ValidationType.COLUMN,
530
+ ValidationType.ROW,
531
+ },
532
+ description=(
533
+ "Which validation types actually count toward a table's/"
534
+ "catalog's overall status and appear in the report. "
535
+ "CATALOG existence and SCHEMA/table discovery always execute "
536
+ "regardless of this setting (discovery is a prerequisite for "
537
+ "COLUMN/ROW checks - there is nothing to compare rows of "
538
+ "without first knowing which tables exist), but their "
539
+ "missing/extra findings are excluded from the status rollup "
540
+ "and report when deselected. COLUMN (name/type/order/"
541
+ "nullable/statistics) and ROW (row count/row-hash) checks are "
542
+ "skipped outright when deselected."
543
+ ),
544
+ )
545
+
546
+ @field_validator("schemas", "tables", "ignore_columns", mode="before")
547
+ @classmethod
548
+ def _ensure_list(cls, value: Any) -> Any:
549
+ if value is None:
550
+ return value
551
+ if isinstance(value, str):
552
+ return [value]
553
+ return list(value)
554
+
555
+ model_config = {"extra": "forbid"}
556
+
557
+
558
+ # ---------------------------------------------------------------------------
559
+ # Azure Blob CSV -> single Databricks table validation request
560
+ #
561
+ # Reuses the exact same result shape (CatalogValidationResponse, wrapping
562
+ # one SchemaValidationResult with one TableValidationResult) as the
563
+ # catalog-to-catalog path above, so report_generator.py needs no changes -
564
+ # it just sees "one schema, one table" instead of many.
565
+ # ---------------------------------------------------------------------------
566
+ class CsvTableValidationRequest(BaseModel):
567
+
568
+ source_blob_path: str = Field(
569
+ ..., min_length=1,
570
+ description="Path to the source CSV inside the Azure Storage container, e.g. 'validation/customers.csv'.",
571
+ )
572
+
573
+ target_catalog: str = Field(..., min_length=1)
574
+ target_schema: str = Field(..., min_length=1)
575
+ target_table: str = Field(..., min_length=1)
576
+
577
+ primary_key: List[str] = Field(
578
+ default_factory=list,
579
+ description=(
580
+ "Primary/business key column(s), used for row-hash and row-level "
581
+ "data comparison. If omitted, comparison falls back to a "
582
+ "synthetic row-number match (CSV file order vs. a Databricks "
583
+ "ROW_NUMBER() over the same column order) - best-effort only, "
584
+ "not a substitute for a real key."
585
+ ),
586
+ )
587
+
588
+ ignore_columns: List[str] = Field(default_factory=list)
589
+
590
+ case_sensitive_columns: bool = Field(default=False)
591
+ validate_column_order: bool = Field(default=True)
592
+
593
+ data_compare_mode: DataCompareMode = Field(
594
+ default_factory=DataCompareMode.default
595
+ )
596
+
597
+ max_sample_rows: int = Field(default=50, ge=1, le=1000)
598
+
599
+ @field_validator("primary_key", "ignore_columns", mode="before")
600
+ @classmethod
601
+ def _ensure_list(cls, value: Any) -> Any:
602
+ if value is None:
603
+ return value
604
+ if isinstance(value, str):
605
+ return [value]
606
+ return list(value)
607
+
608
+ model_config = {"extra": "forbid"}
609
+
610
+
611
+ # ---------------------------------------------------------------------------
612
+ # Azure SQL Database -> Databricks catalog validation request
613
+ #
614
+ # Multi-table, matched by name: every table in the Azure SQL database
615
+ # (optionally restricted to `schemas`/`tables`) is matched against a
616
+ # like-named table in the target Databricks catalog/schema, mirroring
617
+ # CatalogValidationRequest's schema/table matching. Returns the same
618
+ # CatalogValidationResponse shape as CatalogValidator, so
619
+ # report_generator.py needs no changes.
620
+ # ---------------------------------------------------------------------------
621
+ class AzureSqlValidationRequest(BaseModel):
622
+
623
+ target_catalog: str = Field(..., min_length=1)
624
+
625
+ schemas: Optional[List[str]] = Field(
626
+ default=None,
627
+ description=(
628
+ "Restrict validation to these Azure SQL schemas (matched to "
629
+ "same-named Databricks schemas, or remapped via schema_map). "
630
+ "If omitted, all schemas common to both sides are validated."
631
+ ),
632
+ )
633
+
634
+ schema_map: Dict[str, str] = Field(
635
+ default_factory=dict,
636
+ description=(
637
+ "Optional map of Azure SQL schema name -> Databricks schema "
638
+ "name, for when the two sides use different schema names for "
639
+ "the same logical migration target (e.g. Azure SQL's default "
640
+ "'dbo' vs a purpose-named Databricks schema). Unmapped "
641
+ "schemas are matched by identical name as usual."
642
+ ),
643
+ )
644
+
645
+ tables: Optional[List[str]] = Field(
646
+ default=None,
647
+ description="Restrict validation to these table names. If omitted, all common tables are validated.",
648
+ )
649
+
650
+ table_map: Dict[str, str] = Field(
651
+ default_factory=dict,
652
+ description=(
653
+ "Optional map of Azure SQL table name -> Databricks table "
654
+ "name, for when the user has explicitly named a source and "
655
+ "target table that don't share the same name - an explicit "
656
+ "pair like this is compared directly, bypassing name-based "
657
+ "table matching entirely (unlike `tables`, which still "
658
+ "requires the name to appear in the intersection). Unmapped "
659
+ "tables are matched by identical name as usual."
660
+ ),
661
+ )
662
+
663
+ ignore_columns: List[str] = Field(default_factory=list)
664
+
665
+ case_sensitive_columns: bool = Field(default=False)
666
+ validate_column_order: bool = Field(default=True)
667
+
668
+ primary_keys: Dict[str, List[str]] = Field(
669
+ default_factory=dict,
670
+ description=(
671
+ "Map of 'schema.table' -> key column list, used for row-hash "
672
+ "and row-level data comparison. Tables without an entry get "
673
+ "schema/row-count/statistics validation only."
674
+ ),
675
+ )
676
+
677
+ data_compare_mode: DataCompareMode = Field(
678
+ default_factory=DataCompareMode.default
679
+ )
680
+
681
+ max_sample_rows: int = Field(default=50, ge=1, le=1000)
682
+
683
+ @field_validator("schemas", "tables", "ignore_columns", mode="before")
684
+ @classmethod
685
+ def _ensure_list(cls, value: Any) -> Any:
686
+ if value is None:
687
+ return value
688
+ if isinstance(value, str):
689
+ return [value]
690
+ return list(value)
691
+
692
+ model_config = {"extra": "forbid"}
693
+
694
+
695
+ # ---------------------------------------------------------------------------
696
+ # Result building blocks
697
+ # ---------------------------------------------------------------------------
698
+ class ColumnValidationResult(BaseModel):
699
+
700
+ column: str
701
+ status: ValidationStatus
702
+
703
+ source_data_type: Optional[str] = None
704
+ target_data_type: Optional[str] = None
705
+ data_type_status: Optional[ValidationStatus] = None
706
+
707
+ source_nullable: Optional[bool] = None
708
+ target_nullable: Optional[bool] = None
709
+ nullable_status: Optional[ValidationStatus] = None
710
+
711
+ source_null_count: Optional[int] = None
712
+ target_null_count: Optional[int] = None
713
+ null_count_status: Optional[ValidationStatus] = None
714
+
715
+ source_distinct_count: Optional[int] = None
716
+ target_distinct_count: Optional[int] = None
717
+ distinct_count_status: Optional[ValidationStatus] = None
718
+
719
+ source_min: Optional[Any] = None
720
+ source_max: Optional[Any] = None
721
+ target_min: Optional[Any] = None
722
+ target_max: Optional[Any] = None
723
+ min_max_status: Optional[ValidationStatus] = None
724
+
725
+ error: Optional[str] = None
726
+
727
+ model_config = {"extra": "ignore"}
728
+
729
+
730
+ class RowMismatchDetail(BaseModel):
731
+ """
732
+ Per-row, per-column detail for one changed row (matching key, differing
733
+ value) surfaced by stage 15 in HASH/FULL mode. One instance is produced
734
+ per mismatched column within a row - a row with 3 differing columns
735
+ yields 3 RowMismatchDetail entries sharing the same key/row hashes.
736
+ """
737
+
738
+ schema_name: str
739
+ table: str
740
+
741
+ primary_key: Dict[str, Any] = Field(default_factory=dict)
742
+ mismatch_column: str
743
+
744
+ source_value: Optional[Any] = None
745
+ target_value: Optional[Any] = None
746
+
747
+ source_row_hash: Optional[Any] = None
748
+ target_row_hash: Optional[Any] = None
749
+
750
+ verified: bool = Field(
751
+ default=True,
752
+ description=(
753
+ "True when this row was pinpointed via a real configured "
754
+ "primary key (Tier 5's standard path). False when derived "
755
+ "from the ROW_NUMBER() fallback used when no primary key is "
756
+ "configured - 'row N' on the source and target are only the "
757
+ "same logical record if both sides otherwise contain the "
758
+ "same row set; treat these rows as best-effort, not a "
759
+ "confirmed per-record diff."
760
+ ),
761
+ )
762
+
763
+ model_config = {"extra": "ignore"}
764
+
765
+
766
+ class RowHashMismatch(BaseModel):
767
+ """
768
+ One primary-key's outcome from the row-hash comparison stage: either a
769
+ mismatched whole-row hash (key present on both sides, hashes differ),
770
+ or a key present on only one side. Never exposes row position/order -
771
+ the primary key is always the identity used.
772
+ """
773
+
774
+ primary_key: str
775
+ source_hash: str
776
+ target_hash: str
777
+ status: str # MISMATCH | MISSING_IN_TARGET | MISSING_IN_SOURCE | DUPLICATE_KEY
778
+
779
+ partition_bucket: Optional[str] = Field(
780
+ default=None,
781
+ description=(
782
+ "Which partition bucket this mismatch was found in, when the "
783
+ "table was compared via partitioned Tier 4 (see "
784
+ "TableValidationResult.partition_column). None for an "
785
+ "unpartitioned (whole-table) row-hash comparison."
786
+ ),
787
+ )
788
+
789
+ model_config = {"extra": "ignore"}
790
+
791
+
792
+ class DataValidationResult(BaseModel):
793
+ """Result of stage 15 (actual row-level data comparison)."""
794
+
795
+ mode: DataCompareMode
796
+ status: ValidationStatus
797
+
798
+ source_only_rows: Optional[int] = None
799
+ target_only_rows: Optional[int] = None
800
+ changed_rows: Optional[int] = None
801
+
802
+ key_columns: List[str] = Field(default_factory=list)
803
+ sample_source_only: List[Dict[str, Any]] = Field(default_factory=list)
804
+ sample_target_only: List[Dict[str, Any]] = Field(default_factory=list)
805
+ sample_changed: List[Dict[str, Any]] = Field(default_factory=list)
806
+ sample_changed_detail: List[RowMismatchDetail] = Field(default_factory=list)
807
+
808
+ # Row-hash comparison stage (separate mechanism from the EXCEPT/hash-join
809
+ # diff above) - pushed-down, per-key whole-row hash comparison. Primary
810
+ # mechanism for detecting row-level mismatches when a key is configured.
811
+ row_hash_mismatches: List[RowHashMismatch] = Field(default_factory=list)
812
+ row_hash_mismatch_count: int = 0
813
+ row_hash_mismatch_percentage: float = 0.0
814
+
815
+ # Tier 2 whole-table fingerprint (Databricks -> Databricks tiered
816
+ # funnel only). fingerprint_status PASS means Tier 4/5 were skipped
817
+ # because the fingerprint already proved the tables equal.
818
+ fingerprint_status: Optional[ValidationStatus] = None
819
+ source_fingerprint: Optional[str] = None
820
+ target_fingerprint: Optional[str] = None
821
+
822
+ # Tier 4: keys that appear more than once on one side (row-hash diff
823
+ # cannot classify a duplicated key the same way as a unique one).
824
+ duplicate_keys: List[str] = Field(default_factory=list)
825
+
826
+ note: Optional[str] = None
827
+ error: Optional[str] = None
828
+
829
+ model_config = {"extra": "ignore"}
830
+
831
+
832
+ class TableValidationResult(BaseModel):
833
+
834
+ schema_name: str
835
+ table: str
836
+ status: ValidationStatus = ValidationStatus.SKIPPED
837
+
838
+ exists_in_source: bool = True
839
+ exists_in_target: bool = True
840
+
841
+ missing_columns: List[str] = Field(default_factory=list)
842
+ extra_columns: List[str] = Field(default_factory=list)
843
+ columns_status: ValidationStatus = ValidationStatus.SKIPPED
844
+
845
+ column_order_status: ValidationStatus = ValidationStatus.SKIPPED
846
+ source_column_order: List[str] = Field(default_factory=list)
847
+ target_column_order: List[str] = Field(default_factory=list)
848
+
849
+ row_count_source: Optional[int] = None
850
+ row_count_target: Optional[int] = None
851
+ row_count_difference: Optional[int] = None
852
+ row_count_status: ValidationStatus = ValidationStatus.SKIPPED
853
+
854
+ columns: List[ColumnValidationResult] = Field(default_factory=list)
855
+ data_types_status: ValidationStatus = ValidationStatus.SKIPPED
856
+ nullable_status: ValidationStatus = ValidationStatus.SKIPPED
857
+ null_counts_status: ValidationStatus = ValidationStatus.SKIPPED
858
+ distinct_counts_status: ValidationStatus = ValidationStatus.SKIPPED
859
+ min_max_status: ValidationStatus = ValidationStatus.SKIPPED
860
+
861
+ data: Optional[DataValidationResult] = None
862
+
863
+ error: Optional[str] = None
864
+
865
+ # Tiered fail-fast funnel bookkeeping (Databricks -> Databricks path
866
+ # only). tier_reached tells a reader exactly how much work was done
867
+ # to reach this table's verdict; schema_blocking distinguishes an
868
+ # aborted table (no row-level tier ever ran) from one that merely has
869
+ # a non-blocking schema note alongside a real row-level result.
870
+ tier_reached: ValidationTier = ValidationTier.SCHEMA_ONLY
871
+ tier_stop_reason: Optional[str] = None
872
+ schema_blocking: bool = False
873
+
874
+ # Partitioned Tier 4 bookkeeping (Databricks -> Databricks path only).
875
+ # Describes HOW Tier 4 was scoped for a large, confirmed-mismatched
876
+ # table - orthogonal to tier_reached, which still just says how far
877
+ # the funnel went (ROW_HASH/COLUMN_DIFF), partitioned or not.
878
+ partitioned: bool = False
879
+ partition_column: Optional[str] = None
880
+ partition_buckets_total: Optional[int] = None
881
+ partition_buckets_culprit: Optional[int] = None
882
+ partition_skip_reason: Optional[str] = Field(
883
+ default=None,
884
+ description=(
885
+ "Why a large, confirmed-mismatched table was NOT partitioned "
886
+ "(e.g. 'below partition_threshold', 'no partition_prompt "
887
+ "configured', 'user declined', '--yes flag / non-interactive "
888
+ "run'). None when the table was too small to be offered "
889
+ "partitioning at all, or when it was partitioned successfully."
890
+ ),
891
+ )
892
+
893
+ model_config = {"extra": "ignore"}
894
+
895
+
896
+ class SchemaValidationResult(BaseModel):
897
+
898
+ schema_name: str
899
+ status: ValidationStatus
900
+
901
+ exists_in_source: bool = True
902
+ exists_in_target: bool = True
903
+
904
+ missing_tables: List[str] = Field(default_factory=list)
905
+ extra_tables: List[str] = Field(default_factory=list)
906
+
907
+ tables: List[TableValidationResult] = Field(default_factory=list)
908
+
909
+ error: Optional[str] = None
910
+
911
+ model_config = {"extra": "ignore"}
912
+
913
+
914
+ class ValidationSummary(BaseModel):
915
+
916
+ total_schemas: int = 0
917
+ passed_schemas: int = 0
918
+ failed_schemas: int = 0
919
+
920
+ total_tables: int = 0
921
+ passed_tables: int = 0
922
+ failed_tables: int = 0
923
+ error_tables: int = 0
924
+ missing_tables: int = 0
925
+ extra_tables: int = 0
926
+
927
+ model_config = {"extra": "ignore"}
928
+
929
+
930
+ class CatalogValidationResponse(BaseModel):
931
+
932
+ source_catalog: str
933
+ target_catalog: str
934
+ status: ValidationStatus
935
+
936
+ validation_timestamp: Optional[str] = Field(
937
+ default=None,
938
+ description="UTC ISO-8601 timestamp when this validation run started.",
939
+ )
940
+
941
+ execution_time_seconds: float = Field(default=0.0, ge=0.0)
942
+
943
+ missing_schemas: List[str] = Field(default_factory=list)
944
+ extra_schemas: List[str] = Field(default_factory=list)
945
+
946
+ summary: ValidationSummary = Field(default_factory=ValidationSummary)
947
+
948
+ schemas: List[SchemaValidationResult] = Field(default_factory=list)
949
+
950
+ error: Optional[str] = None
951
+
952
+ model_config = {"extra": "ignore"}