table-validator 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- table_validator/__init__.py +46 -0
- table_validator/auth/__init__.py +1 -0
- table_validator/auth/azure_auth.py +52 -0
- table_validator/auth/databricks_auth.py +31 -0
- table_validator/cli/__init__.py +1 -0
- table_validator/cli/main.py +722 -0
- table_validator/cli/partition_prompt.py +78 -0
- table_validator/cli/summary_table.py +146 -0
- table_validator/cli/wizard.py +429 -0
- table_validator/config/__init__.py +1 -0
- table_validator/config/manager.py +84 -0
- table_validator/config/schema.py +179 -0
- table_validator/connectors/__init__.py +1 -0
- table_validator/connectors/azure_connector.py +809 -0
- table_validator/connectors/databricks_connector.py +1230 -0
- table_validator/engine/__init__.py +1 -0
- table_validator/engine/comparison_engine.py +645 -0
- table_validator/models.py +952 -0
- table_validator/reports/__init__.py +1 -0
- table_validator/reports/excel_report.py +953 -0
- table_validator/validators/__init__.py +1 -0
- table_validator/validators/blob_discovery.py +467 -0
- table_validator/validators/catalog_validator.py +1863 -0
- table_validator/validators/row_validator.py +1727 -0
- table_validator-0.1.0.dist-info/METADATA +190 -0
- table_validator-0.1.0.dist-info/RECORD +30 -0
- table_validator-0.1.0.dist-info/WHEEL +5 -0
- table_validator-0.1.0.dist-info/entry_points.txt +2 -0
- table_validator-0.1.0.dist-info/licenses/LICENSE +21 -0
- table_validator-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Validators package: comparison/decision logic for catalogs and row-level data."""
|
|
@@ -0,0 +1,467 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Azure Blob Storage -> Databricks catalog: multi-blob discovery and
|
|
3
|
+
lightweight comparison.
|
|
4
|
+
|
|
5
|
+
Unlike AzureCsvValidator (which validates ONE named blob against ONE named
|
|
6
|
+
Databricks table, with full row-hash/row-level comparison), this module
|
|
7
|
+
answers "which blobs in a container correspond to which tables in a
|
|
8
|
+
Databricks catalog?" - list blobs matching folder_prefix/file_pattern,
|
|
9
|
+
infer a candidate table name from each blob's filename (strip path and
|
|
10
|
+
extension), and intersect those inferred names against the catalog's
|
|
11
|
+
actual table names, using the same list-and-intersect-by-name approach
|
|
12
|
+
CatalogValidator already uses for schema/table discovery.
|
|
13
|
+
|
|
14
|
+
BlobCatalogValidator.validate()'s target_schema is optional, same
|
|
15
|
+
"blank means compare everything" convention as
|
|
16
|
+
CatalogValidator/AzureSqlValidator: if left unset, every schema in
|
|
17
|
+
target_catalog (excluding information_schema) is listed and blobs are
|
|
18
|
+
matched against each in turn, aggregating every schema's results onto
|
|
19
|
+
one CatalogValidationResponse.
|
|
20
|
+
|
|
21
|
+
Comparison for each matched (blob, table) pair is intentionally lighter
|
|
22
|
+
than AzureCsvValidator's: row count, plus column name/type comparison
|
|
23
|
+
only. Row-hash / row-level data-mismatch comparison is deliberately out
|
|
24
|
+
of scope here - AzureCsvValidator's hash-formatting rules
|
|
25
|
+
(_format_value_for_hash) were built and empirically verified against
|
|
26
|
+
CSV-sourced pandas dtypes specifically; extending them to every blob
|
|
27
|
+
format matched by a wildcard pattern (Parquet's native int64 vs CSV's
|
|
28
|
+
string-then-inferred columns, for example) without the same empirical
|
|
29
|
+
verification risks the exact class of silent-wrong-mismatch bug fixed
|
|
30
|
+
earlier for decimal-vs-bigint and NVARCHAR-vs-VARCHAR encoding. A user
|
|
31
|
+
who wants full row-hash comparison against one specific file can still
|
|
32
|
+
use the existing single-blob CsvTableValidationRequest/AzureCsvValidator
|
|
33
|
+
path directly.
|
|
34
|
+
|
|
35
|
+
Column comparison reuses AzureCsvValidator's static helpers
|
|
36
|
+
(_infer_databricks_type, _types_compatible) rather than reimplementing
|
|
37
|
+
them, since those are already correct and don't depend on any per-file
|
|
38
|
+
state.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import datetime
|
|
44
|
+
import logging
|
|
45
|
+
import time
|
|
46
|
+
from typing import List, Optional, Tuple
|
|
47
|
+
|
|
48
|
+
import pandas as pd
|
|
49
|
+
|
|
50
|
+
from table_validator.connectors.azure_connector import AzureConnector
|
|
51
|
+
from table_validator.connectors.databricks_connector import DatabricksConnector
|
|
52
|
+
from table_validator.models import (
|
|
53
|
+
CatalogValidationResponse,
|
|
54
|
+
ColumnValidationResult,
|
|
55
|
+
SchemaValidationResult,
|
|
56
|
+
TableValidationResult,
|
|
57
|
+
ValidationStatus,
|
|
58
|
+
ValidationSummary,
|
|
59
|
+
)
|
|
60
|
+
from table_validator.validators.row_validator import AzureCsvValidator
|
|
61
|
+
|
|
62
|
+
logger = logging.getLogger(__name__)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _infer_table_name(blob_path: str) -> str:
|
|
66
|
+
"""Strip directory path and extension from a blob path to get a
|
|
67
|
+
candidate table name, e.g. 'validation/2024/Customers.csv' -> 'Customers'."""
|
|
68
|
+
base_name = blob_path.rsplit("/", 1)[-1]
|
|
69
|
+
return base_name.rsplit(".", 1)[0] if "." in base_name else base_name
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def discover_blob_table_matches(
|
|
73
|
+
blob_names: List[str],
|
|
74
|
+
catalog_tables: List[str],
|
|
75
|
+
) -> Tuple[List[Tuple[str, str]], List[str], List[str]]:
|
|
76
|
+
"""
|
|
77
|
+
Match blobs to catalog tables by inferred name (case-insensitive),
|
|
78
|
+
mirroring CatalogValidator.compare_tables' intersection-by-name
|
|
79
|
+
approach.
|
|
80
|
+
|
|
81
|
+
Returns (matched_pairs, blob_only, table_only) where matched_pairs is
|
|
82
|
+
a list of (blob_path, table_name) using each side's own real name/
|
|
83
|
+
casing, blob_only is blob paths with no matching table, and
|
|
84
|
+
table_only is catalog table names with no matching blob.
|
|
85
|
+
"""
|
|
86
|
+
blob_by_inferred_name = {_infer_table_name(b).lower(): b for b in blob_names}
|
|
87
|
+
table_by_name = {t.lower(): t for t in catalog_tables}
|
|
88
|
+
|
|
89
|
+
common_keys = set(blob_by_inferred_name) & set(table_by_name)
|
|
90
|
+
blob_only_keys = set(blob_by_inferred_name) - set(table_by_name)
|
|
91
|
+
table_only_keys = set(table_by_name) - set(blob_by_inferred_name)
|
|
92
|
+
|
|
93
|
+
matched_pairs = sorted(
|
|
94
|
+
((blob_by_inferred_name[k], table_by_name[k]) for k in common_keys),
|
|
95
|
+
key=lambda pair: pair[1],
|
|
96
|
+
)
|
|
97
|
+
blob_only = sorted(blob_by_inferred_name[k] for k in blob_only_keys)
|
|
98
|
+
table_only = sorted(table_by_name[k] for k in table_only_keys)
|
|
99
|
+
|
|
100
|
+
return matched_pairs, blob_only, table_only
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class BlobCatalogValidator:
|
|
104
|
+
"""
|
|
105
|
+
Validates every blob in a container (optionally scoped by
|
|
106
|
+
folder_prefix/file_pattern) that matches a same-named table in a
|
|
107
|
+
Databricks catalog schema, by inferred filename. Row count + column
|
|
108
|
+
name/type comparison only (see module docstring for why row-hash
|
|
109
|
+
comparison is out of scope here).
|
|
110
|
+
"""
|
|
111
|
+
|
|
112
|
+
# Databricks-managed system schema, present in every catalog - never a
|
|
113
|
+
# real migration target. Same convention as CatalogValidator.
|
|
114
|
+
_EXCLUDED_SCHEMAS = {"information_schema"}
|
|
115
|
+
|
|
116
|
+
def __init__(
|
|
117
|
+
self,
|
|
118
|
+
azure_connector: AzureConnector,
|
|
119
|
+
databricks_connector: DatabricksConnector,
|
|
120
|
+
) -> None:
|
|
121
|
+
self.azure = azure_connector
|
|
122
|
+
self.databricks = databricks_connector
|
|
123
|
+
logger.debug("BlobCatalogValidator initialised")
|
|
124
|
+
|
|
125
|
+
def validate(
|
|
126
|
+
self,
|
|
127
|
+
target_catalog: str,
|
|
128
|
+
target_schema: Optional[str] = None,
|
|
129
|
+
folder_prefix: Optional[str] = None,
|
|
130
|
+
file_pattern: Optional[str] = None,
|
|
131
|
+
blob_path: Optional[str] = None,
|
|
132
|
+
target_table: Optional[str] = None,
|
|
133
|
+
) -> CatalogValidationResponse:
|
|
134
|
+
"""
|
|
135
|
+
target_schema=None means "match blobs against every schema in
|
|
136
|
+
target_catalog" - lists all schemas (excluding
|
|
137
|
+
information_schema), matches blobs against each in turn, and
|
|
138
|
+
aggregates every schema's results onto one response, same
|
|
139
|
+
"blank means compare everything" convention as
|
|
140
|
+
CatalogValidator/AzureSqlValidator.
|
|
141
|
+
|
|
142
|
+
If both blob_path and target_table are given, filename-to-table
|
|
143
|
+
discovery is bypassed entirely: that exact blob is compared
|
|
144
|
+
directly against that exact table, even if their names don't
|
|
145
|
+
match (same "explicit pair skips name matching" idea as
|
|
146
|
+
AzureSqlValidator's schema_map/table_map). target_schema is
|
|
147
|
+
still required in this mode, since a bare table name alone
|
|
148
|
+
doesn't identify a Databricks schema.
|
|
149
|
+
"""
|
|
150
|
+
start = time.perf_counter()
|
|
151
|
+
run_timestamp = datetime.datetime.now(datetime.timezone.utc).isoformat()
|
|
152
|
+
|
|
153
|
+
if blob_path and target_table:
|
|
154
|
+
if not target_schema:
|
|
155
|
+
return self._error_response(
|
|
156
|
+
target_catalog, run_timestamp, start,
|
|
157
|
+
"target_table.schema is required when both an exact "
|
|
158
|
+
"source blob and target table are given.",
|
|
159
|
+
)
|
|
160
|
+
table_result = self._validate_pair(
|
|
161
|
+
blob_path, target_catalog, target_schema, target_table
|
|
162
|
+
)
|
|
163
|
+
schema_result = SchemaValidationResult(
|
|
164
|
+
schema_name=target_schema,
|
|
165
|
+
status=table_result.status,
|
|
166
|
+
missing_tables=[],
|
|
167
|
+
extra_tables=[],
|
|
168
|
+
tables=[table_result],
|
|
169
|
+
)
|
|
170
|
+
execution_time = round(time.perf_counter() - start, 3)
|
|
171
|
+
logger.info(
|
|
172
|
+
"Blob validation finished (explicit pair) | status=%s | duration=%.3fs",
|
|
173
|
+
table_result.status, execution_time,
|
|
174
|
+
)
|
|
175
|
+
return CatalogValidationResponse(
|
|
176
|
+
source_catalog=f"blob:{self.azure.container_name}",
|
|
177
|
+
target_catalog=target_catalog,
|
|
178
|
+
status=table_result.status,
|
|
179
|
+
validation_timestamp=run_timestamp,
|
|
180
|
+
execution_time_seconds=execution_time,
|
|
181
|
+
summary=ValidationSummary(
|
|
182
|
+
total_schemas=1,
|
|
183
|
+
passed_schemas=1 if table_result.status == ValidationStatus.PASS else 0,
|
|
184
|
+
failed_schemas=0 if table_result.status == ValidationStatus.PASS else 1,
|
|
185
|
+
total_tables=1,
|
|
186
|
+
passed_tables=1 if table_result.status == ValidationStatus.PASS else 0,
|
|
187
|
+
failed_tables=0 if table_result.status == ValidationStatus.PASS else 1,
|
|
188
|
+
error_tables=1 if table_result.status == ValidationStatus.ERROR else 0,
|
|
189
|
+
),
|
|
190
|
+
schemas=[schema_result],
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
try:
|
|
194
|
+
blob_names = self.azure.list_blobs(folder_prefix, file_pattern)
|
|
195
|
+
except Exception as exc:
|
|
196
|
+
logger.exception("Failed to list blobs")
|
|
197
|
+
return self._error_response(
|
|
198
|
+
target_catalog, run_timestamp, start,
|
|
199
|
+
f"Unable to list blobs: {exc}",
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
if target_schema:
|
|
203
|
+
schemas_to_check = [target_schema]
|
|
204
|
+
else:
|
|
205
|
+
try:
|
|
206
|
+
all_schemas = self.databricks.get_schemas(target_catalog)
|
|
207
|
+
except Exception as exc:
|
|
208
|
+
logger.exception("Failed to list schemas for '%s'", target_catalog)
|
|
209
|
+
return self._error_response(
|
|
210
|
+
target_catalog, run_timestamp, start,
|
|
211
|
+
f"Unable to list target schemas: {exc}",
|
|
212
|
+
)
|
|
213
|
+
schemas_to_check = sorted(
|
|
214
|
+
s for s in all_schemas if s.lower() not in self._EXCLUDED_SCHEMAS
|
|
215
|
+
)
|
|
216
|
+
logger.info(
|
|
217
|
+
"No schema configured - matching blobs against all %d schema(s) "
|
|
218
|
+
"in '%s': %s",
|
|
219
|
+
len(schemas_to_check), target_catalog, schemas_to_check,
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
schema_results: List[SchemaValidationResult] = [
|
|
223
|
+
self._validate_schema(blob_names, target_catalog, schema_name)
|
|
224
|
+
for schema_name in schemas_to_check
|
|
225
|
+
]
|
|
226
|
+
|
|
227
|
+
overall_status = _calculate_overall_status([s.status for s in schema_results])
|
|
228
|
+
|
|
229
|
+
summary = ValidationSummary(
|
|
230
|
+
total_schemas=len(schema_results),
|
|
231
|
+
)
|
|
232
|
+
for schema_result in schema_results:
|
|
233
|
+
if schema_result.status == ValidationStatus.PASS:
|
|
234
|
+
summary.passed_schemas += 1
|
|
235
|
+
else:
|
|
236
|
+
summary.failed_schemas += 1
|
|
237
|
+
summary.total_tables += len(schema_result.tables)
|
|
238
|
+
summary.missing_tables += len(schema_result.missing_tables)
|
|
239
|
+
summary.extra_tables += len(schema_result.extra_tables)
|
|
240
|
+
for table in schema_result.tables:
|
|
241
|
+
if table.status == ValidationStatus.PASS:
|
|
242
|
+
summary.passed_tables += 1
|
|
243
|
+
elif table.status == ValidationStatus.ERROR:
|
|
244
|
+
summary.error_tables += 1
|
|
245
|
+
summary.failed_tables += 1
|
|
246
|
+
elif table.status == ValidationStatus.FAIL:
|
|
247
|
+
summary.failed_tables += 1
|
|
248
|
+
|
|
249
|
+
execution_time = round(time.perf_counter() - start, 3)
|
|
250
|
+
logger.info(
|
|
251
|
+
"Blob validation finished | status=%s | duration=%.3fs",
|
|
252
|
+
overall_status, execution_time,
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
return CatalogValidationResponse(
|
|
256
|
+
source_catalog=f"blob:{self.azure.container_name}",
|
|
257
|
+
target_catalog=target_catalog,
|
|
258
|
+
status=overall_status,
|
|
259
|
+
validation_timestamp=run_timestamp,
|
|
260
|
+
execution_time_seconds=execution_time,
|
|
261
|
+
missing_schemas=[],
|
|
262
|
+
extra_schemas=[],
|
|
263
|
+
summary=summary,
|
|
264
|
+
schemas=schema_results,
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
def _validate_schema(
|
|
268
|
+
self,
|
|
269
|
+
blob_names: List[str],
|
|
270
|
+
target_catalog: str,
|
|
271
|
+
target_schema: str,
|
|
272
|
+
) -> SchemaValidationResult:
|
|
273
|
+
try:
|
|
274
|
+
catalog_tables = self.databricks.get_tables(target_catalog, target_schema)
|
|
275
|
+
except Exception as exc:
|
|
276
|
+
logger.exception(
|
|
277
|
+
"Failed to list tables for '%s.%s'", target_catalog, target_schema
|
|
278
|
+
)
|
|
279
|
+
return SchemaValidationResult(
|
|
280
|
+
schema_name=target_schema,
|
|
281
|
+
status=ValidationStatus.ERROR,
|
|
282
|
+
error=f"Unable to list target tables: {exc}",
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
matched_pairs, blob_only, table_only = discover_blob_table_matches(
|
|
286
|
+
blob_names, catalog_tables
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
# Same "missing" / "extra" convention as CatalogValidator/
|
|
290
|
+
# AzureSqlValidator: missing_tables = present in source, absent in
|
|
291
|
+
# target; extra_tables = present in target, absent in source. Here
|
|
292
|
+
# the blob is the source and the Databricks table is the target,
|
|
293
|
+
# so a blob with no matching table is "missing from target"
|
|
294
|
+
# (blob_only), and a table with no matching blob is "extra in
|
|
295
|
+
# target" (table_only) - NOT the other way around.
|
|
296
|
+
if blob_only:
|
|
297
|
+
logger.warning(
|
|
298
|
+
"Blobs with no matching table in '%s.%s' (missing from target): %s",
|
|
299
|
+
target_catalog, target_schema, blob_only,
|
|
300
|
+
)
|
|
301
|
+
if table_only:
|
|
302
|
+
logger.warning(
|
|
303
|
+
"Tables in '%s.%s' with no matching blob (extra in target): %s",
|
|
304
|
+
target_catalog, target_schema, table_only,
|
|
305
|
+
)
|
|
306
|
+
logger.info(
|
|
307
|
+
"Found %d matching blob/table pair(s) in '%s.%s'.",
|
|
308
|
+
len(matched_pairs), target_catalog, target_schema,
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
table_results: List[TableValidationResult] = [
|
|
312
|
+
self._validate_pair(blob_path, target_catalog, target_schema, table_name)
|
|
313
|
+
for blob_path, table_name in matched_pairs
|
|
314
|
+
]
|
|
315
|
+
|
|
316
|
+
statuses = [t.status for t in table_results]
|
|
317
|
+
if blob_only:
|
|
318
|
+
statuses.append(ValidationStatus.FAIL)
|
|
319
|
+
status = _calculate_overall_status(statuses)
|
|
320
|
+
|
|
321
|
+
return SchemaValidationResult(
|
|
322
|
+
schema_name=target_schema,
|
|
323
|
+
status=status,
|
|
324
|
+
missing_tables=blob_only,
|
|
325
|
+
extra_tables=table_only,
|
|
326
|
+
tables=table_results,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
def _error_response(
|
|
330
|
+
self,
|
|
331
|
+
target_catalog: str,
|
|
332
|
+
run_timestamp: str,
|
|
333
|
+
start: float,
|
|
334
|
+
error: str,
|
|
335
|
+
) -> CatalogValidationResponse:
|
|
336
|
+
return CatalogValidationResponse(
|
|
337
|
+
source_catalog=f"blob:{self.azure.container_name}",
|
|
338
|
+
target_catalog=target_catalog,
|
|
339
|
+
status=ValidationStatus.ERROR,
|
|
340
|
+
validation_timestamp=run_timestamp,
|
|
341
|
+
execution_time_seconds=round(time.perf_counter() - start, 3),
|
|
342
|
+
error=error,
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
def _validate_pair(
|
|
346
|
+
self,
|
|
347
|
+
blob_path: str,
|
|
348
|
+
target_catalog: str,
|
|
349
|
+
target_schema: str,
|
|
350
|
+
target_table: str,
|
|
351
|
+
) -> TableValidationResult:
|
|
352
|
+
result = TableValidationResult(schema_name=target_schema, table=target_table)
|
|
353
|
+
|
|
354
|
+
try:
|
|
355
|
+
source_df = self.azure.read_csv(blob_path)
|
|
356
|
+
except Exception as exc:
|
|
357
|
+
logger.exception("Failed to read blob '%s'", blob_path)
|
|
358
|
+
result.status = ValidationStatus.ERROR
|
|
359
|
+
result.error = f"Unable to read blob '{blob_path}': {exc}"
|
|
360
|
+
return result
|
|
361
|
+
|
|
362
|
+
try:
|
|
363
|
+
target_schema_df = self.databricks.get_table_schema(
|
|
364
|
+
target_catalog, target_schema, target_table
|
|
365
|
+
)
|
|
366
|
+
except Exception as exc:
|
|
367
|
+
logger.exception(
|
|
368
|
+
"Failed to retrieve column metadata for '%s.%s.%s'",
|
|
369
|
+
target_catalog, target_schema, target_table,
|
|
370
|
+
)
|
|
371
|
+
result.status = ValidationStatus.ERROR
|
|
372
|
+
result.error = f"Unable to retrieve column metadata: {exc}"
|
|
373
|
+
return result
|
|
374
|
+
|
|
375
|
+
# Column names (common / missing / extra), reusing the same
|
|
376
|
+
# normalization convention as CatalogValidator/AzureCsvValidator.
|
|
377
|
+
src_cols = {str(c).lower(): str(c) for c in source_df.columns}
|
|
378
|
+
tgt_cols = {
|
|
379
|
+
str(c).lower(): str(c) for c in target_schema_df["column_name"]
|
|
380
|
+
}
|
|
381
|
+
missing_cols = sorted(set(src_cols) - set(tgt_cols))
|
|
382
|
+
extra_cols = sorted(set(tgt_cols) - set(src_cols))
|
|
383
|
+
common_cols = sorted(src_cols[k] for k in (set(src_cols) & set(tgt_cols)))
|
|
384
|
+
|
|
385
|
+
result.missing_columns = missing_cols
|
|
386
|
+
result.extra_columns = extra_cols
|
|
387
|
+
result.columns_status = (
|
|
388
|
+
ValidationStatus.FAIL if (missing_cols or extra_cols) else ValidationStatus.PASS
|
|
389
|
+
)
|
|
390
|
+
|
|
391
|
+
if not common_cols:
|
|
392
|
+
result.status = ValidationStatus.FAIL
|
|
393
|
+
result.error = "No common columns between blob and target table"
|
|
394
|
+
return result
|
|
395
|
+
|
|
396
|
+
tgt_type_by_col = {
|
|
397
|
+
str(r["column_name"]).lower(): str(r["data_type"])
|
|
398
|
+
for _, r in target_schema_df.iterrows()
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
column_results: List[ColumnValidationResult] = []
|
|
402
|
+
dtype_statuses = []
|
|
403
|
+
for col in common_cols:
|
|
404
|
+
src_type = AzureCsvValidator._infer_databricks_type(source_df[col])
|
|
405
|
+
tgt_type = tgt_type_by_col.get(col.lower(), "")
|
|
406
|
+
status = (
|
|
407
|
+
ValidationStatus.PASS
|
|
408
|
+
if AzureCsvValidator._types_compatible(src_type, tgt_type)
|
|
409
|
+
else ValidationStatus.FAIL
|
|
410
|
+
)
|
|
411
|
+
dtype_statuses.append(status)
|
|
412
|
+
column_results.append(
|
|
413
|
+
ColumnValidationResult(
|
|
414
|
+
column=col,
|
|
415
|
+
status=status,
|
|
416
|
+
source_data_type=src_type,
|
|
417
|
+
target_data_type=tgt_type,
|
|
418
|
+
data_type_status=status,
|
|
419
|
+
)
|
|
420
|
+
)
|
|
421
|
+
|
|
422
|
+
result.columns = column_results
|
|
423
|
+
result.data_types_status = _calculate_overall_status(dtype_statuses)
|
|
424
|
+
|
|
425
|
+
# Row count only - no row-hash / row-level comparison (see module
|
|
426
|
+
# docstring).
|
|
427
|
+
try:
|
|
428
|
+
src_count = len(source_df)
|
|
429
|
+
tgt_count = self.databricks.get_row_count(
|
|
430
|
+
target_catalog, target_schema, target_table
|
|
431
|
+
)
|
|
432
|
+
result.row_count_source = src_count
|
|
433
|
+
result.row_count_target = tgt_count
|
|
434
|
+
result.row_count_difference = tgt_count - src_count
|
|
435
|
+
result.row_count_status = (
|
|
436
|
+
ValidationStatus.PASS if src_count == tgt_count else ValidationStatus.FAIL
|
|
437
|
+
)
|
|
438
|
+
except Exception as exc:
|
|
439
|
+
logger.exception("Failed to compute row count for '%s'", target_table)
|
|
440
|
+
result.row_count_status = ValidationStatus.ERROR
|
|
441
|
+
result.error = f"Row count failed: {exc}"
|
|
442
|
+
|
|
443
|
+
result.status = _calculate_overall_status(
|
|
444
|
+
[
|
|
445
|
+
result.columns_status,
|
|
446
|
+
result.data_types_status,
|
|
447
|
+
result.row_count_status,
|
|
448
|
+
]
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
return result
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _calculate_overall_status(
|
|
455
|
+
statuses: List[Optional[ValidationStatus]],
|
|
456
|
+
) -> ValidationStatus:
|
|
457
|
+
"""Same precedence rule as CatalogValidator.calculate_overall_status."""
|
|
458
|
+
clean = [s for s in statuses if s is not None]
|
|
459
|
+
if not clean:
|
|
460
|
+
return ValidationStatus.SKIPPED
|
|
461
|
+
if any(s == ValidationStatus.ERROR for s in clean):
|
|
462
|
+
return ValidationStatus.ERROR
|
|
463
|
+
if any(s == ValidationStatus.FAIL for s in clean):
|
|
464
|
+
return ValidationStatus.FAIL
|
|
465
|
+
if all(s == ValidationStatus.SKIPPED for s in clean):
|
|
466
|
+
return ValidationStatus.SKIPPED
|
|
467
|
+
return ValidationStatus.PASS
|