table-validator 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- table_validator/__init__.py +46 -0
- table_validator/auth/__init__.py +1 -0
- table_validator/auth/azure_auth.py +52 -0
- table_validator/auth/databricks_auth.py +31 -0
- table_validator/cli/__init__.py +1 -0
- table_validator/cli/main.py +722 -0
- table_validator/cli/partition_prompt.py +78 -0
- table_validator/cli/summary_table.py +146 -0
- table_validator/cli/wizard.py +429 -0
- table_validator/config/__init__.py +1 -0
- table_validator/config/manager.py +84 -0
- table_validator/config/schema.py +179 -0
- table_validator/connectors/__init__.py +1 -0
- table_validator/connectors/azure_connector.py +809 -0
- table_validator/connectors/databricks_connector.py +1230 -0
- table_validator/engine/__init__.py +1 -0
- table_validator/engine/comparison_engine.py +645 -0
- table_validator/models.py +952 -0
- table_validator/reports/__init__.py +1 -0
- table_validator/reports/excel_report.py +953 -0
- table_validator/validators/__init__.py +1 -0
- table_validator/validators/blob_discovery.py +467 -0
- table_validator/validators/catalog_validator.py +1863 -0
- table_validator/validators/row_validator.py +1727 -0
- table_validator-0.1.0.dist-info/METADATA +190 -0
- table_validator-0.1.0.dist-info/RECORD +30 -0
- table_validator-0.1.0.dist-info/WHEEL +5 -0
- table_validator-0.1.0.dist-info/entry_points.txt +2 -0
- table_validator-0.1.0.dist-info/licenses/LICENSE +21 -0
- table_validator-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,952 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Pydantic models for the Data Migration Comparison Service.
|
|
3
|
+
|
|
4
|
+
Defines request / response contracts used by the FastAPI layer and the
|
|
5
|
+
comparison engine.
|
|
6
|
+
|
|
7
|
+
This file contains two families of models:
|
|
8
|
+
|
|
9
|
+
1. Existing CSV-vs-Databricks row comparison models (CompareRequest /
|
|
10
|
+
ComparisonResult / etc.) - UNCHANGED from the original implementation.
|
|
11
|
+
|
|
12
|
+
2. New Databricks catalog-to-catalog validation models, added to support
|
|
13
|
+
CatalogValidator in comparison_engine.py. These are intentionally kept
|
|
14
|
+
separate (different enum, different result shape) rather than
|
|
15
|
+
shoehorned into the existing ComparisonStatus / ComparisonResult
|
|
16
|
+
models, since a catalog validation run produces a tree of results
|
|
17
|
+
(catalog -> schemas -> tables -> columns) rather than a single flat
|
|
18
|
+
comparison.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from enum import Enum
|
|
24
|
+
from typing import Any, Dict, List, Optional, Set
|
|
25
|
+
|
|
26
|
+
from pydantic import BaseModel, Field, field_validator
|
|
27
|
+
|
|
28
|
+
from table_validator.config.schema import ValidationType
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
# ---------------------------------------------------------------------------
|
|
32
|
+
# Enumerations
|
|
33
|
+
# ---------------------------------------------------------------------------
|
|
34
|
+
class Platform(str, Enum):
|
|
35
|
+
AZURE_STORAGE = "azure_storage"
|
|
36
|
+
DATABRICKS = "databricks"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class ComparisonStatus(str, Enum):
|
|
40
|
+
PASS = "PASS"
|
|
41
|
+
WARN = "WARN"
|
|
42
|
+
FAIL = "FAIL"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# ---------------------------------------------------------------------------
|
|
46
|
+
# Nested configuration models
|
|
47
|
+
# ---------------------------------------------------------------------------
|
|
48
|
+
class SourceConfig(BaseModel):
|
|
49
|
+
|
|
50
|
+
platform: Platform = Field(
|
|
51
|
+
default=Platform.AZURE_STORAGE,
|
|
52
|
+
description="Source platform identifier",
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
table: str = Field(
|
|
56
|
+
...,
|
|
57
|
+
min_length=1,
|
|
58
|
+
description="Source CSV file path inside Azure Storage",
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
query: Optional[str] = Field(
|
|
62
|
+
default=None,
|
|
63
|
+
description="Reserved for future use.",
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
model_config = {"extra": "forbid"}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class TargetConfig(BaseModel):
|
|
70
|
+
|
|
71
|
+
platform: Platform = Field(
|
|
72
|
+
default=Platform.DATABRICKS,
|
|
73
|
+
description="Target platform identifier",
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
table: str = Field(
|
|
77
|
+
...,
|
|
78
|
+
min_length=1,
|
|
79
|
+
description="Target Databricks table name",
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
query: Optional[str] = Field(
|
|
83
|
+
default=None,
|
|
84
|
+
description="Optional Databricks SQL query.",
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
model_config = {"extra": "forbid"}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class ComparisonOptions(BaseModel):
|
|
91
|
+
|
|
92
|
+
primary_keys: List[str] = Field(
|
|
93
|
+
default_factory=list,
|
|
94
|
+
description="Primary key column(s).",
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
ignore_columns: List[str] = Field(
|
|
98
|
+
default_factory=list,
|
|
99
|
+
description="Columns ignored during comparison.",
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
compare_schema: bool = Field(
|
|
103
|
+
default=True,
|
|
104
|
+
description="Enable schema comparison",
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
compare_values: bool = Field(
|
|
108
|
+
default=True,
|
|
109
|
+
description="Enable value comparison",
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
compare_duplicates: bool = Field(
|
|
113
|
+
default=True,
|
|
114
|
+
description="Enable duplicate detection",
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
case_sensitive: bool = Field(
|
|
118
|
+
default=True,
|
|
119
|
+
description="Case-sensitive comparison",
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
trim_strings: bool = Field(
|
|
123
|
+
default=True,
|
|
124
|
+
description="Trim leading/trailing spaces before comparison",
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
numeric_tolerance: float = Field(
|
|
128
|
+
default=0.0,
|
|
129
|
+
ge=0.0,
|
|
130
|
+
description="Numeric comparison tolerance",
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
sample_size: int = Field(
|
|
134
|
+
default=20,
|
|
135
|
+
ge=1,
|
|
136
|
+
le=500,
|
|
137
|
+
description="Maximum mismatch samples returned",
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
model_config = {"extra": "forbid"}
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# ---------------------------------------------------------------------------
|
|
144
|
+
# Request model
|
|
145
|
+
# ---------------------------------------------------------------------------
|
|
146
|
+
class CompareRequest(BaseModel):
|
|
147
|
+
|
|
148
|
+
source_platform: Platform = Field(
|
|
149
|
+
default=Platform.AZURE_STORAGE,
|
|
150
|
+
description="Source platform",
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
target_platform: Platform = Field(
|
|
154
|
+
default=Platform.DATABRICKS,
|
|
155
|
+
description="Target platform",
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
source_table: str = Field(
|
|
159
|
+
...,
|
|
160
|
+
min_length=1,
|
|
161
|
+
description="Azure Storage CSV path (example: n8ndirectory/day.csv)",
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
target_table: str = Field(
|
|
165
|
+
...,
|
|
166
|
+
min_length=1,
|
|
167
|
+
description="Databricks table name",
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
source_query: Optional[str] = Field(
|
|
171
|
+
default=None,
|
|
172
|
+
description="Reserved for future use.",
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
target_query: Optional[str] = Field(
|
|
176
|
+
default=None,
|
|
177
|
+
description="Optional Databricks SQL query.",
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
primary_keys: List[str] = Field(
|
|
181
|
+
default_factory=list,
|
|
182
|
+
description="Primary key column(s)",
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
ignore_columns: List[str] = Field(
|
|
186
|
+
default_factory=list,
|
|
187
|
+
description="Columns ignored during comparison",
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
compare_schema: bool = Field(default=True)
|
|
191
|
+
compare_values: bool = Field(default=True)
|
|
192
|
+
compare_duplicates: bool = Field(default=True)
|
|
193
|
+
case_sensitive: bool = Field(default=True)
|
|
194
|
+
|
|
195
|
+
trim_strings: bool = Field(
|
|
196
|
+
default=True,
|
|
197
|
+
description="Trim whitespace before comparison",
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
numeric_tolerance: float = Field(
|
|
201
|
+
default=0.0,
|
|
202
|
+
ge=0.0,
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
sample_size: int = Field(
|
|
206
|
+
default=20,
|
|
207
|
+
ge=1,
|
|
208
|
+
le=500,
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
@classmethod
|
|
212
|
+
def from_configs(
|
|
213
|
+
cls,
|
|
214
|
+
source: SourceConfig,
|
|
215
|
+
target: TargetConfig,
|
|
216
|
+
options: Optional[ComparisonOptions] = None,
|
|
217
|
+
) -> "CompareRequest":
|
|
218
|
+
opts = options or ComparisonOptions()
|
|
219
|
+
return cls(
|
|
220
|
+
source_platform=source.platform,
|
|
221
|
+
target_platform=target.platform,
|
|
222
|
+
source_table=source.table,
|
|
223
|
+
target_table=target.table,
|
|
224
|
+
source_query=source.query,
|
|
225
|
+
target_query=target.query,
|
|
226
|
+
primary_keys=opts.primary_keys,
|
|
227
|
+
ignore_columns=opts.ignore_columns,
|
|
228
|
+
compare_schema=opts.compare_schema,
|
|
229
|
+
compare_values=opts.compare_values,
|
|
230
|
+
compare_duplicates=opts.compare_duplicates,
|
|
231
|
+
case_sensitive=opts.case_sensitive,
|
|
232
|
+
trim_strings=opts.trim_strings,
|
|
233
|
+
numeric_tolerance=opts.numeric_tolerance,
|
|
234
|
+
sample_size=opts.sample_size,
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
@field_validator("primary_keys", "ignore_columns", mode="before")
|
|
238
|
+
@classmethod
|
|
239
|
+
def _ensure_list(cls, value: Any) -> List[str]:
|
|
240
|
+
if value is None:
|
|
241
|
+
return []
|
|
242
|
+
if isinstance(value, str):
|
|
243
|
+
return [value]
|
|
244
|
+
return list(value)
|
|
245
|
+
|
|
246
|
+
model_config = {
|
|
247
|
+
"extra": "forbid",
|
|
248
|
+
"json_schema_extra": {
|
|
249
|
+
"example": {
|
|
250
|
+
"source_platform": "azure_storage",
|
|
251
|
+
"target_platform": "databricks",
|
|
252
|
+
"source_table": "n8ndirectory/day.csv",
|
|
253
|
+
"target_table": "for_n8n_catalog.for_n8n_scheme.day",
|
|
254
|
+
"primary_keys": ["Numeric"],
|
|
255
|
+
"ignore_columns": [],
|
|
256
|
+
"compare_schema": True,
|
|
257
|
+
"compare_values": True,
|
|
258
|
+
"compare_duplicates": True,
|
|
259
|
+
"case_sensitive": False,
|
|
260
|
+
"trim_strings": True,
|
|
261
|
+
"numeric_tolerance": 0.01,
|
|
262
|
+
"sample_size": 20,
|
|
263
|
+
}
|
|
264
|
+
},
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
# ---------------------------------------------------------------------------
|
|
269
|
+
# Response models
|
|
270
|
+
# ---------------------------------------------------------------------------
|
|
271
|
+
class ComparisonResult(BaseModel):
|
|
272
|
+
|
|
273
|
+
status: ComparisonStatus = Field(
|
|
274
|
+
...,
|
|
275
|
+
description="Overall comparison outcome",
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
execution_time: float = Field(
|
|
279
|
+
...,
|
|
280
|
+
alias="execution_time_seconds",
|
|
281
|
+
ge=0.0,
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
row_count_source: int = Field(..., ge=0)
|
|
285
|
+
row_count_target: int = Field(..., ge=0)
|
|
286
|
+
|
|
287
|
+
matched_rows: int = Field(..., ge=0)
|
|
288
|
+
missing_rows: int = Field(..., ge=0)
|
|
289
|
+
extra_rows: int = Field(..., ge=0)
|
|
290
|
+
|
|
291
|
+
duplicate_rows: Dict[str, Any] = Field(
|
|
292
|
+
default_factory=dict,
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
schema_match: bool
|
|
296
|
+
|
|
297
|
+
column_differences: List[Dict[str, Any]] = Field(
|
|
298
|
+
default_factory=list,
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
sample_mismatches: List[Dict[str, Any]] = Field(
|
|
302
|
+
default_factory=list,
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
model_config = {
|
|
306
|
+
"populate_by_name": True,
|
|
307
|
+
"extra": "ignore",
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
class HealthResponse(BaseModel):
|
|
312
|
+
|
|
313
|
+
status: str = Field(
|
|
314
|
+
default="healthy",
|
|
315
|
+
examples=["healthy"],
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
model_config = {"extra": "forbid"}
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
# ---------------------------------------------------------------------------
|
|
322
|
+
# Backward compatibility aliases
|
|
323
|
+
# ---------------------------------------------------------------------------
|
|
324
|
+
ComparisonRequest = CompareRequest
|
|
325
|
+
ComparisonResponse = ComparisonResult
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
# ===========================================================================
|
|
329
|
+
# NEW: Databricks catalog-to-catalog validation models
|
|
330
|
+
# ===========================================================================
|
|
331
|
+
class ValidationStatus(str, Enum):
|
|
332
|
+
"""
|
|
333
|
+
Status for a single validation stage or an aggregated object
|
|
334
|
+
(column / table / schema / catalog).
|
|
335
|
+
|
|
336
|
+
Distinct from ComparisonStatus (PASS/WARN/FAIL) because the catalog
|
|
337
|
+
validator needs to separate genuine validation failures from
|
|
338
|
+
technical errors (permission denied, connection dropped, etc.) and
|
|
339
|
+
from stages that were intentionally skipped by configuration.
|
|
340
|
+
"""
|
|
341
|
+
|
|
342
|
+
PASS = "PASS"
|
|
343
|
+
FAIL = "FAIL"
|
|
344
|
+
ERROR = "ERROR"
|
|
345
|
+
SKIPPED = "SKIPPED"
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
class DataCompareMode(str, Enum):
|
|
349
|
+
"""
|
|
350
|
+
Controls how expensive stage 15 (actual row-level data comparison) is.
|
|
351
|
+
Everything runs as push-down SQL against Databricks; no full-table
|
|
352
|
+
collect() / toPandas() is ever performed regardless of mode.
|
|
353
|
+
"""
|
|
354
|
+
|
|
355
|
+
COUNT_ONLY = "COUNT_ONLY" # row count only, skip null/distinct/minmax/data
|
|
356
|
+
STATISTICS = "STATISTICS" # null/distinct/minmax, skip row-level data compare
|
|
357
|
+
HASH = "HASH" # key + row-hash based row compare (pushed down)
|
|
358
|
+
FULL = "FULL" # key-based anti-join, returns sample mismatched rows
|
|
359
|
+
|
|
360
|
+
@classmethod
|
|
361
|
+
def default(cls) -> "DataCompareMode":
|
|
362
|
+
# Safe default for large datasets: aggregate statistics, no
|
|
363
|
+
# row-level comparison.
|
|
364
|
+
return cls.STATISTICS
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
class ValidationTier(int, Enum):
|
|
368
|
+
"""
|
|
369
|
+
How far the tiered fail-fast funnel got for one table (Databricks ->
|
|
370
|
+
Databricks path only, see CatalogValidator). Each tier only runs if
|
|
371
|
+
every cheaper tier before it failed to conclusively answer "are these
|
|
372
|
+
tables equal" - a table's tier_reached tells a reader exactly how much
|
|
373
|
+
work was (and, just as importantly, was NOT) done to reach its
|
|
374
|
+
verdict, which the previous "everything always runs" pipeline had no
|
|
375
|
+
way to express.
|
|
376
|
+
|
|
377
|
+
Tier 3 (partition/bucket fingerprinting) is deferred to a follow-up;
|
|
378
|
+
a mismatch at Tier 2 goes straight to Tier 4 over the whole table.
|
|
379
|
+
"""
|
|
380
|
+
|
|
381
|
+
SCHEMA_BLOCKED = 0 # Tier 0 found a BLOCKING schema diff; aborted, no further tier ran
|
|
382
|
+
SCHEMA_ONLY = 1 # Tier 0 was the final word (e.g. ROW validation disabled)
|
|
383
|
+
STATISTICAL = 2 # Tier 1 was the final word (mismatch, or max_tier capped here)
|
|
384
|
+
FINGERPRINT = 3 # Tier 2 was the final word (whole-table fingerprint matched)
|
|
385
|
+
ROW_HASH = 4 # Tier 4 ran (per-key row-hash diff)
|
|
386
|
+
COLUMN_DIFF = 5 # Tier 5 ran (column-level diff of mismatched keys)
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
class HashCanonicalizationSpec(BaseModel):
|
|
390
|
+
"""
|
|
391
|
+
Canonicalization rules shared by every hashing tier (Tier 2 whole-table
|
|
392
|
+
fingerprint, Tier 4 per-key row hash) so they can never disagree with
|
|
393
|
+
each other about what "the same row" hashes to.
|
|
394
|
+
|
|
395
|
+
Defaults reproduce the pre-existing, empirically-verified hash
|
|
396
|
+
expression byte-for-byte (see databricks_connector.get_row_hashes'
|
|
397
|
+
history and table_validator/CLAUDE.md's note on sha2()/hashlib.sha256
|
|
398
|
+
cross-dialect equivalence) - nothing changes until a caller
|
|
399
|
+
deliberately constructs a non-default spec. Not yet exposed via CLI/
|
|
400
|
+
wizard: these knobs are correctness-sensitive and need dedicated
|
|
401
|
+
verification against real data before users can toggle them.
|
|
402
|
+
"""
|
|
403
|
+
|
|
404
|
+
null_sentinel: str = "\x01NULL\x01"
|
|
405
|
+
trim_strings: bool = False
|
|
406
|
+
case_sensitive: bool = True
|
|
407
|
+
float_rounding_decimals: Optional[int] = None
|
|
408
|
+
normalize_negative_zero: bool = False
|
|
409
|
+
unicode_nfc_normalize: bool = False
|
|
410
|
+
|
|
411
|
+
model_config = {"extra": "forbid"}
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
class PartitionPromptContext(BaseModel):
|
|
415
|
+
"""
|
|
416
|
+
Passed to CatalogValidator's optional partition_prompt callback when a
|
|
417
|
+
table is both large (row count over request.partition_threshold) and
|
|
418
|
+
has a confirmed mismatch (Tier 1 and/or Tier 2) - i.e. exactly the
|
|
419
|
+
case where an unpartitioned Tier 4 row-hash diff would be expensive
|
|
420
|
+
and a bucketed comparison is worth offering. The callback returns the
|
|
421
|
+
chosen partition column name, or None to decline (falls back to
|
|
422
|
+
today's unpartitioned Tier 4 over the whole table).
|
|
423
|
+
|
|
424
|
+
Kept as a plain data-carrier with no behavior, so the validator (pure
|
|
425
|
+
decision logic, no I/O) and the CLI (the only place that actually
|
|
426
|
+
prompts a human) stay cleanly separated - the validator only ever
|
|
427
|
+
calls this callback and reads its return value.
|
|
428
|
+
"""
|
|
429
|
+
|
|
430
|
+
schema_name: str
|
|
431
|
+
table: str
|
|
432
|
+
row_count: int
|
|
433
|
+
candidate_columns: List[str]
|
|
434
|
+
|
|
435
|
+
model_config = {"extra": "forbid"}
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
# ---------------------------------------------------------------------------
|
|
439
|
+
# Request
|
|
440
|
+
# ---------------------------------------------------------------------------
|
|
441
|
+
class CatalogValidationRequest(BaseModel):
|
|
442
|
+
|
|
443
|
+
source_catalog: str = Field(..., min_length=1)
|
|
444
|
+
target_catalog: str = Field(..., min_length=1)
|
|
445
|
+
|
|
446
|
+
schemas: Optional[List[str]] = Field(
|
|
447
|
+
default=None,
|
|
448
|
+
description=(
|
|
449
|
+
"Restrict validation to these schemas. If omitted, all "
|
|
450
|
+
"schemas common to both catalogs are validated."
|
|
451
|
+
),
|
|
452
|
+
)
|
|
453
|
+
|
|
454
|
+
tables: Optional[List[str]] = Field(
|
|
455
|
+
default=None,
|
|
456
|
+
description=(
|
|
457
|
+
"Restrict validation to these table names (applies within "
|
|
458
|
+
"every validated schema). If omitted, all common tables are "
|
|
459
|
+
"validated."
|
|
460
|
+
),
|
|
461
|
+
)
|
|
462
|
+
|
|
463
|
+
ignore_columns: List[str] = Field(default_factory=list)
|
|
464
|
+
|
|
465
|
+
case_sensitive_columns: bool = Field(
|
|
466
|
+
default=False,
|
|
467
|
+
description="Case-sensitive column name comparison.",
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
validate_column_order: bool = Field(
|
|
471
|
+
default=True,
|
|
472
|
+
description="If False, column order differences do not fail a table.",
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
validate_nullable: bool = Field(default=True)
|
|
476
|
+
|
|
477
|
+
primary_keys: Dict[str, List[str]] = Field(
|
|
478
|
+
default_factory=dict,
|
|
479
|
+
description=(
|
|
480
|
+
"Optional map of 'schema.table' -> key column list, used for "
|
|
481
|
+
"row-level data comparison (HASH / FULL modes). Tables without "
|
|
482
|
+
"an entry fall back to a safe COUNT_ONLY-style comparison for "
|
|
483
|
+
"stage 15."
|
|
484
|
+
),
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
data_compare_mode: DataCompareMode = Field(
|
|
488
|
+
default_factory=DataCompareMode.default
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
max_tier: ValidationTier = Field(
|
|
492
|
+
default=ValidationTier.COLUMN_DIFF,
|
|
493
|
+
description=(
|
|
494
|
+
"Ceiling for the tiered fail-fast funnel (Tier 0 schema -> "
|
|
495
|
+
"Tier 1 statistics -> Tier 2 fingerprint -> Tier 4 row-hash -> "
|
|
496
|
+
"Tier 5 column diff). STATISTICAL stops after Tier 1 even if "
|
|
497
|
+
"it matches (the --mode=stats CLI flag); COLUMN_DIFF (default) "
|
|
498
|
+
"lets the funnel run all the way to a column-level diff if "
|
|
499
|
+
"cheaper tiers can't already prove the tables equal. Only "
|
|
500
|
+
"takes effect when ROW validation is enabled - it is inert "
|
|
501
|
+
"otherwise, same as data_compare_mode was."
|
|
502
|
+
),
|
|
503
|
+
)
|
|
504
|
+
|
|
505
|
+
max_sample_rows: int = Field(
|
|
506
|
+
default=50,
|
|
507
|
+
ge=1,
|
|
508
|
+
le=1000,
|
|
509
|
+
description="Max sample mismatched rows returned per table in FULL mode.",
|
|
510
|
+
)
|
|
511
|
+
|
|
512
|
+
partition_threshold: int = Field(
|
|
513
|
+
default=1_000_000,
|
|
514
|
+
ge=1,
|
|
515
|
+
description=(
|
|
516
|
+
"Row-count threshold above which a table with a confirmed "
|
|
517
|
+
"mismatch (Tier 1 and/or Tier 2) is offered for partitioned "
|
|
518
|
+
"Tier 4 row-hash comparison via the partition_prompt callback, "
|
|
519
|
+
"instead of an unpartitioned whole-table scan. Below this "
|
|
520
|
+
"threshold, or when partition_prompt is not supplied/declines, "
|
|
521
|
+
"Tier 4 always runs unpartitioned as before."
|
|
522
|
+
),
|
|
523
|
+
)
|
|
524
|
+
|
|
525
|
+
enabled_validations: Set[ValidationType] = Field(
|
|
526
|
+
default_factory=lambda: {
|
|
527
|
+
ValidationType.CATALOG,
|
|
528
|
+
ValidationType.SCHEMA,
|
|
529
|
+
ValidationType.COLUMN,
|
|
530
|
+
ValidationType.ROW,
|
|
531
|
+
},
|
|
532
|
+
description=(
|
|
533
|
+
"Which validation types actually count toward a table's/"
|
|
534
|
+
"catalog's overall status and appear in the report. "
|
|
535
|
+
"CATALOG existence and SCHEMA/table discovery always execute "
|
|
536
|
+
"regardless of this setting (discovery is a prerequisite for "
|
|
537
|
+
"COLUMN/ROW checks - there is nothing to compare rows of "
|
|
538
|
+
"without first knowing which tables exist), but their "
|
|
539
|
+
"missing/extra findings are excluded from the status rollup "
|
|
540
|
+
"and report when deselected. COLUMN (name/type/order/"
|
|
541
|
+
"nullable/statistics) and ROW (row count/row-hash) checks are "
|
|
542
|
+
"skipped outright when deselected."
|
|
543
|
+
),
|
|
544
|
+
)
|
|
545
|
+
|
|
546
|
+
@field_validator("schemas", "tables", "ignore_columns", mode="before")
|
|
547
|
+
@classmethod
|
|
548
|
+
def _ensure_list(cls, value: Any) -> Any:
|
|
549
|
+
if value is None:
|
|
550
|
+
return value
|
|
551
|
+
if isinstance(value, str):
|
|
552
|
+
return [value]
|
|
553
|
+
return list(value)
|
|
554
|
+
|
|
555
|
+
model_config = {"extra": "forbid"}
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
# ---------------------------------------------------------------------------
|
|
559
|
+
# Azure Blob CSV -> single Databricks table validation request
|
|
560
|
+
#
|
|
561
|
+
# Reuses the exact same result shape (CatalogValidationResponse, wrapping
|
|
562
|
+
# one SchemaValidationResult with one TableValidationResult) as the
|
|
563
|
+
# catalog-to-catalog path above, so report_generator.py needs no changes -
|
|
564
|
+
# it just sees "one schema, one table" instead of many.
|
|
565
|
+
# ---------------------------------------------------------------------------
|
|
566
|
+
class CsvTableValidationRequest(BaseModel):
|
|
567
|
+
|
|
568
|
+
source_blob_path: str = Field(
|
|
569
|
+
..., min_length=1,
|
|
570
|
+
description="Path to the source CSV inside the Azure Storage container, e.g. 'validation/customers.csv'.",
|
|
571
|
+
)
|
|
572
|
+
|
|
573
|
+
target_catalog: str = Field(..., min_length=1)
|
|
574
|
+
target_schema: str = Field(..., min_length=1)
|
|
575
|
+
target_table: str = Field(..., min_length=1)
|
|
576
|
+
|
|
577
|
+
primary_key: List[str] = Field(
|
|
578
|
+
default_factory=list,
|
|
579
|
+
description=(
|
|
580
|
+
"Primary/business key column(s), used for row-hash and row-level "
|
|
581
|
+
"data comparison. If omitted, comparison falls back to a "
|
|
582
|
+
"synthetic row-number match (CSV file order vs. a Databricks "
|
|
583
|
+
"ROW_NUMBER() over the same column order) - best-effort only, "
|
|
584
|
+
"not a substitute for a real key."
|
|
585
|
+
),
|
|
586
|
+
)
|
|
587
|
+
|
|
588
|
+
ignore_columns: List[str] = Field(default_factory=list)
|
|
589
|
+
|
|
590
|
+
case_sensitive_columns: bool = Field(default=False)
|
|
591
|
+
validate_column_order: bool = Field(default=True)
|
|
592
|
+
|
|
593
|
+
data_compare_mode: DataCompareMode = Field(
|
|
594
|
+
default_factory=DataCompareMode.default
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
max_sample_rows: int = Field(default=50, ge=1, le=1000)
|
|
598
|
+
|
|
599
|
+
@field_validator("primary_key", "ignore_columns", mode="before")
|
|
600
|
+
@classmethod
|
|
601
|
+
def _ensure_list(cls, value: Any) -> Any:
|
|
602
|
+
if value is None:
|
|
603
|
+
return value
|
|
604
|
+
if isinstance(value, str):
|
|
605
|
+
return [value]
|
|
606
|
+
return list(value)
|
|
607
|
+
|
|
608
|
+
model_config = {"extra": "forbid"}
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
# ---------------------------------------------------------------------------
|
|
612
|
+
# Azure SQL Database -> Databricks catalog validation request
|
|
613
|
+
#
|
|
614
|
+
# Multi-table, matched by name: every table in the Azure SQL database
|
|
615
|
+
# (optionally restricted to `schemas`/`tables`) is matched against a
|
|
616
|
+
# like-named table in the target Databricks catalog/schema, mirroring
|
|
617
|
+
# CatalogValidationRequest's schema/table matching. Returns the same
|
|
618
|
+
# CatalogValidationResponse shape as CatalogValidator, so
|
|
619
|
+
# report_generator.py needs no changes.
|
|
620
|
+
# ---------------------------------------------------------------------------
|
|
621
|
+
class AzureSqlValidationRequest(BaseModel):
|
|
622
|
+
|
|
623
|
+
target_catalog: str = Field(..., min_length=1)
|
|
624
|
+
|
|
625
|
+
schemas: Optional[List[str]] = Field(
|
|
626
|
+
default=None,
|
|
627
|
+
description=(
|
|
628
|
+
"Restrict validation to these Azure SQL schemas (matched to "
|
|
629
|
+
"same-named Databricks schemas, or remapped via schema_map). "
|
|
630
|
+
"If omitted, all schemas common to both sides are validated."
|
|
631
|
+
),
|
|
632
|
+
)
|
|
633
|
+
|
|
634
|
+
schema_map: Dict[str, str] = Field(
|
|
635
|
+
default_factory=dict,
|
|
636
|
+
description=(
|
|
637
|
+
"Optional map of Azure SQL schema name -> Databricks schema "
|
|
638
|
+
"name, for when the two sides use different schema names for "
|
|
639
|
+
"the same logical migration target (e.g. Azure SQL's default "
|
|
640
|
+
"'dbo' vs a purpose-named Databricks schema). Unmapped "
|
|
641
|
+
"schemas are matched by identical name as usual."
|
|
642
|
+
),
|
|
643
|
+
)
|
|
644
|
+
|
|
645
|
+
tables: Optional[List[str]] = Field(
|
|
646
|
+
default=None,
|
|
647
|
+
description="Restrict validation to these table names. If omitted, all common tables are validated.",
|
|
648
|
+
)
|
|
649
|
+
|
|
650
|
+
table_map: Dict[str, str] = Field(
|
|
651
|
+
default_factory=dict,
|
|
652
|
+
description=(
|
|
653
|
+
"Optional map of Azure SQL table name -> Databricks table "
|
|
654
|
+
"name, for when the user has explicitly named a source and "
|
|
655
|
+
"target table that don't share the same name - an explicit "
|
|
656
|
+
"pair like this is compared directly, bypassing name-based "
|
|
657
|
+
"table matching entirely (unlike `tables`, which still "
|
|
658
|
+
"requires the name to appear in the intersection). Unmapped "
|
|
659
|
+
"tables are matched by identical name as usual."
|
|
660
|
+
),
|
|
661
|
+
)
|
|
662
|
+
|
|
663
|
+
ignore_columns: List[str] = Field(default_factory=list)
|
|
664
|
+
|
|
665
|
+
case_sensitive_columns: bool = Field(default=False)
|
|
666
|
+
validate_column_order: bool = Field(default=True)
|
|
667
|
+
|
|
668
|
+
primary_keys: Dict[str, List[str]] = Field(
|
|
669
|
+
default_factory=dict,
|
|
670
|
+
description=(
|
|
671
|
+
"Map of 'schema.table' -> key column list, used for row-hash "
|
|
672
|
+
"and row-level data comparison. Tables without an entry get "
|
|
673
|
+
"schema/row-count/statistics validation only."
|
|
674
|
+
),
|
|
675
|
+
)
|
|
676
|
+
|
|
677
|
+
data_compare_mode: DataCompareMode = Field(
|
|
678
|
+
default_factory=DataCompareMode.default
|
|
679
|
+
)
|
|
680
|
+
|
|
681
|
+
max_sample_rows: int = Field(default=50, ge=1, le=1000)
|
|
682
|
+
|
|
683
|
+
@field_validator("schemas", "tables", "ignore_columns", mode="before")
|
|
684
|
+
@classmethod
|
|
685
|
+
def _ensure_list(cls, value: Any) -> Any:
|
|
686
|
+
if value is None:
|
|
687
|
+
return value
|
|
688
|
+
if isinstance(value, str):
|
|
689
|
+
return [value]
|
|
690
|
+
return list(value)
|
|
691
|
+
|
|
692
|
+
model_config = {"extra": "forbid"}
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
# ---------------------------------------------------------------------------
|
|
696
|
+
# Result building blocks
|
|
697
|
+
# ---------------------------------------------------------------------------
|
|
698
|
+
class ColumnValidationResult(BaseModel):
|
|
699
|
+
|
|
700
|
+
column: str
|
|
701
|
+
status: ValidationStatus
|
|
702
|
+
|
|
703
|
+
source_data_type: Optional[str] = None
|
|
704
|
+
target_data_type: Optional[str] = None
|
|
705
|
+
data_type_status: Optional[ValidationStatus] = None
|
|
706
|
+
|
|
707
|
+
source_nullable: Optional[bool] = None
|
|
708
|
+
target_nullable: Optional[bool] = None
|
|
709
|
+
nullable_status: Optional[ValidationStatus] = None
|
|
710
|
+
|
|
711
|
+
source_null_count: Optional[int] = None
|
|
712
|
+
target_null_count: Optional[int] = None
|
|
713
|
+
null_count_status: Optional[ValidationStatus] = None
|
|
714
|
+
|
|
715
|
+
source_distinct_count: Optional[int] = None
|
|
716
|
+
target_distinct_count: Optional[int] = None
|
|
717
|
+
distinct_count_status: Optional[ValidationStatus] = None
|
|
718
|
+
|
|
719
|
+
source_min: Optional[Any] = None
|
|
720
|
+
source_max: Optional[Any] = None
|
|
721
|
+
target_min: Optional[Any] = None
|
|
722
|
+
target_max: Optional[Any] = None
|
|
723
|
+
min_max_status: Optional[ValidationStatus] = None
|
|
724
|
+
|
|
725
|
+
error: Optional[str] = None
|
|
726
|
+
|
|
727
|
+
model_config = {"extra": "ignore"}
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
class RowMismatchDetail(BaseModel):
|
|
731
|
+
"""
|
|
732
|
+
Per-row, per-column detail for one changed row (matching key, differing
|
|
733
|
+
value) surfaced by stage 15 in HASH/FULL mode. One instance is produced
|
|
734
|
+
per mismatched column within a row - a row with 3 differing columns
|
|
735
|
+
yields 3 RowMismatchDetail entries sharing the same key/row hashes.
|
|
736
|
+
"""
|
|
737
|
+
|
|
738
|
+
schema_name: str
|
|
739
|
+
table: str
|
|
740
|
+
|
|
741
|
+
primary_key: Dict[str, Any] = Field(default_factory=dict)
|
|
742
|
+
mismatch_column: str
|
|
743
|
+
|
|
744
|
+
source_value: Optional[Any] = None
|
|
745
|
+
target_value: Optional[Any] = None
|
|
746
|
+
|
|
747
|
+
source_row_hash: Optional[Any] = None
|
|
748
|
+
target_row_hash: Optional[Any] = None
|
|
749
|
+
|
|
750
|
+
verified: bool = Field(
|
|
751
|
+
default=True,
|
|
752
|
+
description=(
|
|
753
|
+
"True when this row was pinpointed via a real configured "
|
|
754
|
+
"primary key (Tier 5's standard path). False when derived "
|
|
755
|
+
"from the ROW_NUMBER() fallback used when no primary key is "
|
|
756
|
+
"configured - 'row N' on the source and target are only the "
|
|
757
|
+
"same logical record if both sides otherwise contain the "
|
|
758
|
+
"same row set; treat these rows as best-effort, not a "
|
|
759
|
+
"confirmed per-record diff."
|
|
760
|
+
),
|
|
761
|
+
)
|
|
762
|
+
|
|
763
|
+
model_config = {"extra": "ignore"}
|
|
764
|
+
|
|
765
|
+
|
|
766
|
+
class RowHashMismatch(BaseModel):
|
|
767
|
+
"""
|
|
768
|
+
One primary-key's outcome from the row-hash comparison stage: either a
|
|
769
|
+
mismatched whole-row hash (key present on both sides, hashes differ),
|
|
770
|
+
or a key present on only one side. Never exposes row position/order -
|
|
771
|
+
the primary key is always the identity used.
|
|
772
|
+
"""
|
|
773
|
+
|
|
774
|
+
primary_key: str
|
|
775
|
+
source_hash: str
|
|
776
|
+
target_hash: str
|
|
777
|
+
status: str # MISMATCH | MISSING_IN_TARGET | MISSING_IN_SOURCE | DUPLICATE_KEY
|
|
778
|
+
|
|
779
|
+
partition_bucket: Optional[str] = Field(
|
|
780
|
+
default=None,
|
|
781
|
+
description=(
|
|
782
|
+
"Which partition bucket this mismatch was found in, when the "
|
|
783
|
+
"table was compared via partitioned Tier 4 (see "
|
|
784
|
+
"TableValidationResult.partition_column). None for an "
|
|
785
|
+
"unpartitioned (whole-table) row-hash comparison."
|
|
786
|
+
),
|
|
787
|
+
)
|
|
788
|
+
|
|
789
|
+
model_config = {"extra": "ignore"}
|
|
790
|
+
|
|
791
|
+
|
|
792
|
+
class DataValidationResult(BaseModel):
|
|
793
|
+
"""Result of stage 15 (actual row-level data comparison)."""
|
|
794
|
+
|
|
795
|
+
mode: DataCompareMode
|
|
796
|
+
status: ValidationStatus
|
|
797
|
+
|
|
798
|
+
source_only_rows: Optional[int] = None
|
|
799
|
+
target_only_rows: Optional[int] = None
|
|
800
|
+
changed_rows: Optional[int] = None
|
|
801
|
+
|
|
802
|
+
key_columns: List[str] = Field(default_factory=list)
|
|
803
|
+
sample_source_only: List[Dict[str, Any]] = Field(default_factory=list)
|
|
804
|
+
sample_target_only: List[Dict[str, Any]] = Field(default_factory=list)
|
|
805
|
+
sample_changed: List[Dict[str, Any]] = Field(default_factory=list)
|
|
806
|
+
sample_changed_detail: List[RowMismatchDetail] = Field(default_factory=list)
|
|
807
|
+
|
|
808
|
+
# Row-hash comparison stage (separate mechanism from the EXCEPT/hash-join
|
|
809
|
+
# diff above) - pushed-down, per-key whole-row hash comparison. Primary
|
|
810
|
+
# mechanism for detecting row-level mismatches when a key is configured.
|
|
811
|
+
row_hash_mismatches: List[RowHashMismatch] = Field(default_factory=list)
|
|
812
|
+
row_hash_mismatch_count: int = 0
|
|
813
|
+
row_hash_mismatch_percentage: float = 0.0
|
|
814
|
+
|
|
815
|
+
# Tier 2 whole-table fingerprint (Databricks -> Databricks tiered
|
|
816
|
+
# funnel only). fingerprint_status PASS means Tier 4/5 were skipped
|
|
817
|
+
# because the fingerprint already proved the tables equal.
|
|
818
|
+
fingerprint_status: Optional[ValidationStatus] = None
|
|
819
|
+
source_fingerprint: Optional[str] = None
|
|
820
|
+
target_fingerprint: Optional[str] = None
|
|
821
|
+
|
|
822
|
+
# Tier 4: keys that appear more than once on one side (row-hash diff
|
|
823
|
+
# cannot classify a duplicated key the same way as a unique one).
|
|
824
|
+
duplicate_keys: List[str] = Field(default_factory=list)
|
|
825
|
+
|
|
826
|
+
note: Optional[str] = None
|
|
827
|
+
error: Optional[str] = None
|
|
828
|
+
|
|
829
|
+
model_config = {"extra": "ignore"}
|
|
830
|
+
|
|
831
|
+
|
|
832
|
+
class TableValidationResult(BaseModel):
|
|
833
|
+
|
|
834
|
+
schema_name: str
|
|
835
|
+
table: str
|
|
836
|
+
status: ValidationStatus = ValidationStatus.SKIPPED
|
|
837
|
+
|
|
838
|
+
exists_in_source: bool = True
|
|
839
|
+
exists_in_target: bool = True
|
|
840
|
+
|
|
841
|
+
missing_columns: List[str] = Field(default_factory=list)
|
|
842
|
+
extra_columns: List[str] = Field(default_factory=list)
|
|
843
|
+
columns_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
844
|
+
|
|
845
|
+
column_order_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
846
|
+
source_column_order: List[str] = Field(default_factory=list)
|
|
847
|
+
target_column_order: List[str] = Field(default_factory=list)
|
|
848
|
+
|
|
849
|
+
row_count_source: Optional[int] = None
|
|
850
|
+
row_count_target: Optional[int] = None
|
|
851
|
+
row_count_difference: Optional[int] = None
|
|
852
|
+
row_count_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
853
|
+
|
|
854
|
+
columns: List[ColumnValidationResult] = Field(default_factory=list)
|
|
855
|
+
data_types_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
856
|
+
nullable_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
857
|
+
null_counts_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
858
|
+
distinct_counts_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
859
|
+
min_max_status: ValidationStatus = ValidationStatus.SKIPPED
|
|
860
|
+
|
|
861
|
+
data: Optional[DataValidationResult] = None
|
|
862
|
+
|
|
863
|
+
error: Optional[str] = None
|
|
864
|
+
|
|
865
|
+
# Tiered fail-fast funnel bookkeeping (Databricks -> Databricks path
|
|
866
|
+
# only). tier_reached tells a reader exactly how much work was done
|
|
867
|
+
# to reach this table's verdict; schema_blocking distinguishes an
|
|
868
|
+
# aborted table (no row-level tier ever ran) from one that merely has
|
|
869
|
+
# a non-blocking schema note alongside a real row-level result.
|
|
870
|
+
tier_reached: ValidationTier = ValidationTier.SCHEMA_ONLY
|
|
871
|
+
tier_stop_reason: Optional[str] = None
|
|
872
|
+
schema_blocking: bool = False
|
|
873
|
+
|
|
874
|
+
# Partitioned Tier 4 bookkeeping (Databricks -> Databricks path only).
|
|
875
|
+
# Describes HOW Tier 4 was scoped for a large, confirmed-mismatched
|
|
876
|
+
# table - orthogonal to tier_reached, which still just says how far
|
|
877
|
+
# the funnel went (ROW_HASH/COLUMN_DIFF), partitioned or not.
|
|
878
|
+
partitioned: bool = False
|
|
879
|
+
partition_column: Optional[str] = None
|
|
880
|
+
partition_buckets_total: Optional[int] = None
|
|
881
|
+
partition_buckets_culprit: Optional[int] = None
|
|
882
|
+
partition_skip_reason: Optional[str] = Field(
|
|
883
|
+
default=None,
|
|
884
|
+
description=(
|
|
885
|
+
"Why a large, confirmed-mismatched table was NOT partitioned "
|
|
886
|
+
"(e.g. 'below partition_threshold', 'no partition_prompt "
|
|
887
|
+
"configured', 'user declined', '--yes flag / non-interactive "
|
|
888
|
+
"run'). None when the table was too small to be offered "
|
|
889
|
+
"partitioning at all, or when it was partitioned successfully."
|
|
890
|
+
),
|
|
891
|
+
)
|
|
892
|
+
|
|
893
|
+
model_config = {"extra": "ignore"}
|
|
894
|
+
|
|
895
|
+
|
|
896
|
+
class SchemaValidationResult(BaseModel):
|
|
897
|
+
|
|
898
|
+
schema_name: str
|
|
899
|
+
status: ValidationStatus
|
|
900
|
+
|
|
901
|
+
exists_in_source: bool = True
|
|
902
|
+
exists_in_target: bool = True
|
|
903
|
+
|
|
904
|
+
missing_tables: List[str] = Field(default_factory=list)
|
|
905
|
+
extra_tables: List[str] = Field(default_factory=list)
|
|
906
|
+
|
|
907
|
+
tables: List[TableValidationResult] = Field(default_factory=list)
|
|
908
|
+
|
|
909
|
+
error: Optional[str] = None
|
|
910
|
+
|
|
911
|
+
model_config = {"extra": "ignore"}
|
|
912
|
+
|
|
913
|
+
|
|
914
|
+
class ValidationSummary(BaseModel):
|
|
915
|
+
|
|
916
|
+
total_schemas: int = 0
|
|
917
|
+
passed_schemas: int = 0
|
|
918
|
+
failed_schemas: int = 0
|
|
919
|
+
|
|
920
|
+
total_tables: int = 0
|
|
921
|
+
passed_tables: int = 0
|
|
922
|
+
failed_tables: int = 0
|
|
923
|
+
error_tables: int = 0
|
|
924
|
+
missing_tables: int = 0
|
|
925
|
+
extra_tables: int = 0
|
|
926
|
+
|
|
927
|
+
model_config = {"extra": "ignore"}
|
|
928
|
+
|
|
929
|
+
|
|
930
|
+
class CatalogValidationResponse(BaseModel):
|
|
931
|
+
|
|
932
|
+
source_catalog: str
|
|
933
|
+
target_catalog: str
|
|
934
|
+
status: ValidationStatus
|
|
935
|
+
|
|
936
|
+
validation_timestamp: Optional[str] = Field(
|
|
937
|
+
default=None,
|
|
938
|
+
description="UTC ISO-8601 timestamp when this validation run started.",
|
|
939
|
+
)
|
|
940
|
+
|
|
941
|
+
execution_time_seconds: float = Field(default=0.0, ge=0.0)
|
|
942
|
+
|
|
943
|
+
missing_schemas: List[str] = Field(default_factory=list)
|
|
944
|
+
extra_schemas: List[str] = Field(default_factory=list)
|
|
945
|
+
|
|
946
|
+
summary: ValidationSummary = Field(default_factory=ValidationSummary)
|
|
947
|
+
|
|
948
|
+
schemas: List[SchemaValidationResult] = Field(default_factory=list)
|
|
949
|
+
|
|
950
|
+
error: Optional[str] = None
|
|
951
|
+
|
|
952
|
+
model_config = {"extra": "ignore"}
|