table-validator 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ """Validators package: comparison/decision logic for catalogs and row-level data."""
@@ -0,0 +1,467 @@
1
+ """
2
+ Azure Blob Storage -> Databricks catalog: multi-blob discovery and
3
+ lightweight comparison.
4
+
5
+ Unlike AzureCsvValidator (which validates ONE named blob against ONE named
6
+ Databricks table, with full row-hash/row-level comparison), this module
7
+ answers "which blobs in a container correspond to which tables in a
8
+ Databricks catalog?" - list blobs matching folder_prefix/file_pattern,
9
+ infer a candidate table name from each blob's filename (strip path and
10
+ extension), and intersect those inferred names against the catalog's
11
+ actual table names, using the same list-and-intersect-by-name approach
12
+ CatalogValidator already uses for schema/table discovery.
13
+
14
+ BlobCatalogValidator.validate()'s target_schema is optional, same
15
+ "blank means compare everything" convention as
16
+ CatalogValidator/AzureSqlValidator: if left unset, every schema in
17
+ target_catalog (excluding information_schema) is listed and blobs are
18
+ matched against each in turn, aggregating every schema's results onto
19
+ one CatalogValidationResponse.
20
+
21
+ Comparison for each matched (blob, table) pair is intentionally lighter
22
+ than AzureCsvValidator's: row count, plus column name/type comparison
23
+ only. Row-hash / row-level data-mismatch comparison is deliberately out
24
+ of scope here - AzureCsvValidator's hash-formatting rules
25
+ (_format_value_for_hash) were built and empirically verified against
26
+ CSV-sourced pandas dtypes specifically; extending them to every blob
27
+ format matched by a wildcard pattern (Parquet's native int64 vs CSV's
28
+ string-then-inferred columns, for example) without the same empirical
29
+ verification risks the exact class of silent-wrong-mismatch bug fixed
30
+ earlier for decimal-vs-bigint and NVARCHAR-vs-VARCHAR encoding. A user
31
+ who wants full row-hash comparison against one specific file can still
32
+ use the existing single-blob CsvTableValidationRequest/AzureCsvValidator
33
+ path directly.
34
+
35
+ Column comparison reuses AzureCsvValidator's static helpers
36
+ (_infer_databricks_type, _types_compatible) rather than reimplementing
37
+ them, since those are already correct and don't depend on any per-file
38
+ state.
39
+ """
40
+
41
+ from __future__ import annotations
42
+
43
+ import datetime
44
+ import logging
45
+ import time
46
+ from typing import List, Optional, Tuple
47
+
48
+ import pandas as pd
49
+
50
+ from table_validator.connectors.azure_connector import AzureConnector
51
+ from table_validator.connectors.databricks_connector import DatabricksConnector
52
+ from table_validator.models import (
53
+ CatalogValidationResponse,
54
+ ColumnValidationResult,
55
+ SchemaValidationResult,
56
+ TableValidationResult,
57
+ ValidationStatus,
58
+ ValidationSummary,
59
+ )
60
+ from table_validator.validators.row_validator import AzureCsvValidator
61
+
62
+ logger = logging.getLogger(__name__)
63
+
64
+
65
+ def _infer_table_name(blob_path: str) -> str:
66
+ """Strip directory path and extension from a blob path to get a
67
+ candidate table name, e.g. 'validation/2024/Customers.csv' -> 'Customers'."""
68
+ base_name = blob_path.rsplit("/", 1)[-1]
69
+ return base_name.rsplit(".", 1)[0] if "." in base_name else base_name
70
+
71
+
72
+ def discover_blob_table_matches(
73
+ blob_names: List[str],
74
+ catalog_tables: List[str],
75
+ ) -> Tuple[List[Tuple[str, str]], List[str], List[str]]:
76
+ """
77
+ Match blobs to catalog tables by inferred name (case-insensitive),
78
+ mirroring CatalogValidator.compare_tables' intersection-by-name
79
+ approach.
80
+
81
+ Returns (matched_pairs, blob_only, table_only) where matched_pairs is
82
+ a list of (blob_path, table_name) using each side's own real name/
83
+ casing, blob_only is blob paths with no matching table, and
84
+ table_only is catalog table names with no matching blob.
85
+ """
86
+ blob_by_inferred_name = {_infer_table_name(b).lower(): b for b in blob_names}
87
+ table_by_name = {t.lower(): t for t in catalog_tables}
88
+
89
+ common_keys = set(blob_by_inferred_name) & set(table_by_name)
90
+ blob_only_keys = set(blob_by_inferred_name) - set(table_by_name)
91
+ table_only_keys = set(table_by_name) - set(blob_by_inferred_name)
92
+
93
+ matched_pairs = sorted(
94
+ ((blob_by_inferred_name[k], table_by_name[k]) for k in common_keys),
95
+ key=lambda pair: pair[1],
96
+ )
97
+ blob_only = sorted(blob_by_inferred_name[k] for k in blob_only_keys)
98
+ table_only = sorted(table_by_name[k] for k in table_only_keys)
99
+
100
+ return matched_pairs, blob_only, table_only
101
+
102
+
103
+ class BlobCatalogValidator:
104
+ """
105
+ Validates every blob in a container (optionally scoped by
106
+ folder_prefix/file_pattern) that matches a same-named table in a
107
+ Databricks catalog schema, by inferred filename. Row count + column
108
+ name/type comparison only (see module docstring for why row-hash
109
+ comparison is out of scope here).
110
+ """
111
+
112
+ # Databricks-managed system schema, present in every catalog - never a
113
+ # real migration target. Same convention as CatalogValidator.
114
+ _EXCLUDED_SCHEMAS = {"information_schema"}
115
+
116
+ def __init__(
117
+ self,
118
+ azure_connector: AzureConnector,
119
+ databricks_connector: DatabricksConnector,
120
+ ) -> None:
121
+ self.azure = azure_connector
122
+ self.databricks = databricks_connector
123
+ logger.debug("BlobCatalogValidator initialised")
124
+
125
+ def validate(
126
+ self,
127
+ target_catalog: str,
128
+ target_schema: Optional[str] = None,
129
+ folder_prefix: Optional[str] = None,
130
+ file_pattern: Optional[str] = None,
131
+ blob_path: Optional[str] = None,
132
+ target_table: Optional[str] = None,
133
+ ) -> CatalogValidationResponse:
134
+ """
135
+ target_schema=None means "match blobs against every schema in
136
+ target_catalog" - lists all schemas (excluding
137
+ information_schema), matches blobs against each in turn, and
138
+ aggregates every schema's results onto one response, same
139
+ "blank means compare everything" convention as
140
+ CatalogValidator/AzureSqlValidator.
141
+
142
+ If both blob_path and target_table are given, filename-to-table
143
+ discovery is bypassed entirely: that exact blob is compared
144
+ directly against that exact table, even if their names don't
145
+ match (same "explicit pair skips name matching" idea as
146
+ AzureSqlValidator's schema_map/table_map). target_schema is
147
+ still required in this mode, since a bare table name alone
148
+ doesn't identify a Databricks schema.
149
+ """
150
+ start = time.perf_counter()
151
+ run_timestamp = datetime.datetime.now(datetime.timezone.utc).isoformat()
152
+
153
+ if blob_path and target_table:
154
+ if not target_schema:
155
+ return self._error_response(
156
+ target_catalog, run_timestamp, start,
157
+ "target_table.schema is required when both an exact "
158
+ "source blob and target table are given.",
159
+ )
160
+ table_result = self._validate_pair(
161
+ blob_path, target_catalog, target_schema, target_table
162
+ )
163
+ schema_result = SchemaValidationResult(
164
+ schema_name=target_schema,
165
+ status=table_result.status,
166
+ missing_tables=[],
167
+ extra_tables=[],
168
+ tables=[table_result],
169
+ )
170
+ execution_time = round(time.perf_counter() - start, 3)
171
+ logger.info(
172
+ "Blob validation finished (explicit pair) | status=%s | duration=%.3fs",
173
+ table_result.status, execution_time,
174
+ )
175
+ return CatalogValidationResponse(
176
+ source_catalog=f"blob:{self.azure.container_name}",
177
+ target_catalog=target_catalog,
178
+ status=table_result.status,
179
+ validation_timestamp=run_timestamp,
180
+ execution_time_seconds=execution_time,
181
+ summary=ValidationSummary(
182
+ total_schemas=1,
183
+ passed_schemas=1 if table_result.status == ValidationStatus.PASS else 0,
184
+ failed_schemas=0 if table_result.status == ValidationStatus.PASS else 1,
185
+ total_tables=1,
186
+ passed_tables=1 if table_result.status == ValidationStatus.PASS else 0,
187
+ failed_tables=0 if table_result.status == ValidationStatus.PASS else 1,
188
+ error_tables=1 if table_result.status == ValidationStatus.ERROR else 0,
189
+ ),
190
+ schemas=[schema_result],
191
+ )
192
+
193
+ try:
194
+ blob_names = self.azure.list_blobs(folder_prefix, file_pattern)
195
+ except Exception as exc:
196
+ logger.exception("Failed to list blobs")
197
+ return self._error_response(
198
+ target_catalog, run_timestamp, start,
199
+ f"Unable to list blobs: {exc}",
200
+ )
201
+
202
+ if target_schema:
203
+ schemas_to_check = [target_schema]
204
+ else:
205
+ try:
206
+ all_schemas = self.databricks.get_schemas(target_catalog)
207
+ except Exception as exc:
208
+ logger.exception("Failed to list schemas for '%s'", target_catalog)
209
+ return self._error_response(
210
+ target_catalog, run_timestamp, start,
211
+ f"Unable to list target schemas: {exc}",
212
+ )
213
+ schemas_to_check = sorted(
214
+ s for s in all_schemas if s.lower() not in self._EXCLUDED_SCHEMAS
215
+ )
216
+ logger.info(
217
+ "No schema configured - matching blobs against all %d schema(s) "
218
+ "in '%s': %s",
219
+ len(schemas_to_check), target_catalog, schemas_to_check,
220
+ )
221
+
222
+ schema_results: List[SchemaValidationResult] = [
223
+ self._validate_schema(blob_names, target_catalog, schema_name)
224
+ for schema_name in schemas_to_check
225
+ ]
226
+
227
+ overall_status = _calculate_overall_status([s.status for s in schema_results])
228
+
229
+ summary = ValidationSummary(
230
+ total_schemas=len(schema_results),
231
+ )
232
+ for schema_result in schema_results:
233
+ if schema_result.status == ValidationStatus.PASS:
234
+ summary.passed_schemas += 1
235
+ else:
236
+ summary.failed_schemas += 1
237
+ summary.total_tables += len(schema_result.tables)
238
+ summary.missing_tables += len(schema_result.missing_tables)
239
+ summary.extra_tables += len(schema_result.extra_tables)
240
+ for table in schema_result.tables:
241
+ if table.status == ValidationStatus.PASS:
242
+ summary.passed_tables += 1
243
+ elif table.status == ValidationStatus.ERROR:
244
+ summary.error_tables += 1
245
+ summary.failed_tables += 1
246
+ elif table.status == ValidationStatus.FAIL:
247
+ summary.failed_tables += 1
248
+
249
+ execution_time = round(time.perf_counter() - start, 3)
250
+ logger.info(
251
+ "Blob validation finished | status=%s | duration=%.3fs",
252
+ overall_status, execution_time,
253
+ )
254
+
255
+ return CatalogValidationResponse(
256
+ source_catalog=f"blob:{self.azure.container_name}",
257
+ target_catalog=target_catalog,
258
+ status=overall_status,
259
+ validation_timestamp=run_timestamp,
260
+ execution_time_seconds=execution_time,
261
+ missing_schemas=[],
262
+ extra_schemas=[],
263
+ summary=summary,
264
+ schemas=schema_results,
265
+ )
266
+
267
+ def _validate_schema(
268
+ self,
269
+ blob_names: List[str],
270
+ target_catalog: str,
271
+ target_schema: str,
272
+ ) -> SchemaValidationResult:
273
+ try:
274
+ catalog_tables = self.databricks.get_tables(target_catalog, target_schema)
275
+ except Exception as exc:
276
+ logger.exception(
277
+ "Failed to list tables for '%s.%s'", target_catalog, target_schema
278
+ )
279
+ return SchemaValidationResult(
280
+ schema_name=target_schema,
281
+ status=ValidationStatus.ERROR,
282
+ error=f"Unable to list target tables: {exc}",
283
+ )
284
+
285
+ matched_pairs, blob_only, table_only = discover_blob_table_matches(
286
+ blob_names, catalog_tables
287
+ )
288
+
289
+ # Same "missing" / "extra" convention as CatalogValidator/
290
+ # AzureSqlValidator: missing_tables = present in source, absent in
291
+ # target; extra_tables = present in target, absent in source. Here
292
+ # the blob is the source and the Databricks table is the target,
293
+ # so a blob with no matching table is "missing from target"
294
+ # (blob_only), and a table with no matching blob is "extra in
295
+ # target" (table_only) - NOT the other way around.
296
+ if blob_only:
297
+ logger.warning(
298
+ "Blobs with no matching table in '%s.%s' (missing from target): %s",
299
+ target_catalog, target_schema, blob_only,
300
+ )
301
+ if table_only:
302
+ logger.warning(
303
+ "Tables in '%s.%s' with no matching blob (extra in target): %s",
304
+ target_catalog, target_schema, table_only,
305
+ )
306
+ logger.info(
307
+ "Found %d matching blob/table pair(s) in '%s.%s'.",
308
+ len(matched_pairs), target_catalog, target_schema,
309
+ )
310
+
311
+ table_results: List[TableValidationResult] = [
312
+ self._validate_pair(blob_path, target_catalog, target_schema, table_name)
313
+ for blob_path, table_name in matched_pairs
314
+ ]
315
+
316
+ statuses = [t.status for t in table_results]
317
+ if blob_only:
318
+ statuses.append(ValidationStatus.FAIL)
319
+ status = _calculate_overall_status(statuses)
320
+
321
+ return SchemaValidationResult(
322
+ schema_name=target_schema,
323
+ status=status,
324
+ missing_tables=blob_only,
325
+ extra_tables=table_only,
326
+ tables=table_results,
327
+ )
328
+
329
+ def _error_response(
330
+ self,
331
+ target_catalog: str,
332
+ run_timestamp: str,
333
+ start: float,
334
+ error: str,
335
+ ) -> CatalogValidationResponse:
336
+ return CatalogValidationResponse(
337
+ source_catalog=f"blob:{self.azure.container_name}",
338
+ target_catalog=target_catalog,
339
+ status=ValidationStatus.ERROR,
340
+ validation_timestamp=run_timestamp,
341
+ execution_time_seconds=round(time.perf_counter() - start, 3),
342
+ error=error,
343
+ )
344
+
345
+ def _validate_pair(
346
+ self,
347
+ blob_path: str,
348
+ target_catalog: str,
349
+ target_schema: str,
350
+ target_table: str,
351
+ ) -> TableValidationResult:
352
+ result = TableValidationResult(schema_name=target_schema, table=target_table)
353
+
354
+ try:
355
+ source_df = self.azure.read_csv(blob_path)
356
+ except Exception as exc:
357
+ logger.exception("Failed to read blob '%s'", blob_path)
358
+ result.status = ValidationStatus.ERROR
359
+ result.error = f"Unable to read blob '{blob_path}': {exc}"
360
+ return result
361
+
362
+ try:
363
+ target_schema_df = self.databricks.get_table_schema(
364
+ target_catalog, target_schema, target_table
365
+ )
366
+ except Exception as exc:
367
+ logger.exception(
368
+ "Failed to retrieve column metadata for '%s.%s.%s'",
369
+ target_catalog, target_schema, target_table,
370
+ )
371
+ result.status = ValidationStatus.ERROR
372
+ result.error = f"Unable to retrieve column metadata: {exc}"
373
+ return result
374
+
375
+ # Column names (common / missing / extra), reusing the same
376
+ # normalization convention as CatalogValidator/AzureCsvValidator.
377
+ src_cols = {str(c).lower(): str(c) for c in source_df.columns}
378
+ tgt_cols = {
379
+ str(c).lower(): str(c) for c in target_schema_df["column_name"]
380
+ }
381
+ missing_cols = sorted(set(src_cols) - set(tgt_cols))
382
+ extra_cols = sorted(set(tgt_cols) - set(src_cols))
383
+ common_cols = sorted(src_cols[k] for k in (set(src_cols) & set(tgt_cols)))
384
+
385
+ result.missing_columns = missing_cols
386
+ result.extra_columns = extra_cols
387
+ result.columns_status = (
388
+ ValidationStatus.FAIL if (missing_cols or extra_cols) else ValidationStatus.PASS
389
+ )
390
+
391
+ if not common_cols:
392
+ result.status = ValidationStatus.FAIL
393
+ result.error = "No common columns between blob and target table"
394
+ return result
395
+
396
+ tgt_type_by_col = {
397
+ str(r["column_name"]).lower(): str(r["data_type"])
398
+ for _, r in target_schema_df.iterrows()
399
+ }
400
+
401
+ column_results: List[ColumnValidationResult] = []
402
+ dtype_statuses = []
403
+ for col in common_cols:
404
+ src_type = AzureCsvValidator._infer_databricks_type(source_df[col])
405
+ tgt_type = tgt_type_by_col.get(col.lower(), "")
406
+ status = (
407
+ ValidationStatus.PASS
408
+ if AzureCsvValidator._types_compatible(src_type, tgt_type)
409
+ else ValidationStatus.FAIL
410
+ )
411
+ dtype_statuses.append(status)
412
+ column_results.append(
413
+ ColumnValidationResult(
414
+ column=col,
415
+ status=status,
416
+ source_data_type=src_type,
417
+ target_data_type=tgt_type,
418
+ data_type_status=status,
419
+ )
420
+ )
421
+
422
+ result.columns = column_results
423
+ result.data_types_status = _calculate_overall_status(dtype_statuses)
424
+
425
+ # Row count only - no row-hash / row-level comparison (see module
426
+ # docstring).
427
+ try:
428
+ src_count = len(source_df)
429
+ tgt_count = self.databricks.get_row_count(
430
+ target_catalog, target_schema, target_table
431
+ )
432
+ result.row_count_source = src_count
433
+ result.row_count_target = tgt_count
434
+ result.row_count_difference = tgt_count - src_count
435
+ result.row_count_status = (
436
+ ValidationStatus.PASS if src_count == tgt_count else ValidationStatus.FAIL
437
+ )
438
+ except Exception as exc:
439
+ logger.exception("Failed to compute row count for '%s'", target_table)
440
+ result.row_count_status = ValidationStatus.ERROR
441
+ result.error = f"Row count failed: {exc}"
442
+
443
+ result.status = _calculate_overall_status(
444
+ [
445
+ result.columns_status,
446
+ result.data_types_status,
447
+ result.row_count_status,
448
+ ]
449
+ )
450
+
451
+ return result
452
+
453
+
454
+ def _calculate_overall_status(
455
+ statuses: List[Optional[ValidationStatus]],
456
+ ) -> ValidationStatus:
457
+ """Same precedence rule as CatalogValidator.calculate_overall_status."""
458
+ clean = [s for s in statuses if s is not None]
459
+ if not clean:
460
+ return ValidationStatus.SKIPPED
461
+ if any(s == ValidationStatus.ERROR for s in clean):
462
+ return ValidationStatus.ERROR
463
+ if any(s == ValidationStatus.FAIL for s in clean):
464
+ return ValidationStatus.FAIL
465
+ if all(s == ValidationStatus.SKIPPED for s in clean):
466
+ return ValidationStatus.SKIPPED
467
+ return ValidationStatus.PASS