table-validator 0.1.22__tar.gz → 0.1.23__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {table_validator-0.1.22/table_validator.egg-info → table_validator-0.1.23}/PKG-INFO +2 -1
- {table_validator-0.1.22 → table_validator-0.1.23}/pyproject.toml +4 -0
- {table_validator-0.1.22 → table_validator-0.1.23/table_validator.egg-info}/PKG-INFO +2 -1
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/SOURCES.txt +1 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/requires.txt +1 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/scm_file_list.json +1 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/scm_version.json +2 -2
- table_validator-0.1.23/tests/test_end_to_end_real_engine.py +333 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/LICENSE +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/README.md +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/setup.cfg +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/auth/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/auth/azure_auth.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/auth/databricks_auth.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/main.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/partition_prompt.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/summary_table.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/wizard.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/config/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/config/manager.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/config/schema.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/azure_connector.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/base_sql_connector.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/databricks_connector.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/spark_connector.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/engine/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/engine/comparison_engine.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/models.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/notebook.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/reports/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/reports/excel_report.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/blob_discovery.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/catalog_validator.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/mismatch_classifier.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/row_validator.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/dependency_links.txt +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/entry_points.txt +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/top_level.txt +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/__init__.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_base_sql_connector.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_blob_discovery.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_catalog_validator.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_cli.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_databricks_connector.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_excel_report.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_mismatch_classifier.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_notebook.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_package_import_without_pyspark.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_partition_prompt.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_report_command.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_row_validator.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_spark_connector.py +0 -0
- {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_wizard.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: table-validator
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.23
|
|
4
4
|
Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
|
|
5
5
|
License: MIT
|
|
6
6
|
Requires-Python: >=3.9
|
|
@@ -24,6 +24,7 @@ Provides-Extra: dev
|
|
|
24
24
|
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
25
25
|
Requires-Dist: ruff>=0.4; extra == "dev"
|
|
26
26
|
Requires-Dist: black>=24.0; extra == "dev"
|
|
27
|
+
Requires-Dist: duckdb>=1.0; extra == "dev"
|
|
27
28
|
Provides-Extra: spark
|
|
28
29
|
Requires-Dist: pyspark>=3.4; extra == "spark"
|
|
29
30
|
Dynamic: license-file
|
|
@@ -31,6 +31,10 @@ dev = [
|
|
|
31
31
|
"pytest>=7.0",
|
|
32
32
|
"ruff>=0.4",
|
|
33
33
|
"black>=24.0",
|
|
34
|
+
# tests/test_end_to_end_real_engine.py runs the validator against a
|
|
35
|
+
# real SQL engine rather than mocked query results, so that every
|
|
36
|
+
# user-facing option is proven to actually change what gets validated.
|
|
37
|
+
"duckdb>=1.0",
|
|
34
38
|
]
|
|
35
39
|
# Only needed to construct SparkConnector/call validate_tables() OUTSIDE a
|
|
36
40
|
# real Databricks notebook (e.g. local dev/testing) - inside an actual
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: table-validator
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.23
|
|
4
4
|
Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
|
|
5
5
|
License: MIT
|
|
6
6
|
Requires-Python: >=3.9
|
|
@@ -24,6 +24,7 @@ Provides-Extra: dev
|
|
|
24
24
|
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
25
25
|
Requires-Dist: ruff>=0.4; extra == "dev"
|
|
26
26
|
Requires-Dist: black>=24.0; extra == "dev"
|
|
27
|
+
Requires-Dist: duckdb>=1.0; extra == "dev"
|
|
27
28
|
Provides-Extra: spark
|
|
28
29
|
Requires-Dist: pyspark>=3.4; extra == "spark"
|
|
29
30
|
Dynamic: license-file
|
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
"""End-to-end validation against a REAL SQL engine, not mocked results.
|
|
2
|
+
|
|
3
|
+
Every other test file mocks the connector's query results, which proves
|
|
4
|
+
the validator's own logic but not that a user-facing option actually
|
|
5
|
+
changes what gets validated. These tests create real tables in DuckDB,
|
|
6
|
+
run the real validator against them, and assert on real verdicts - so
|
|
7
|
+
each option offered by the CLI wizard, the validate_tables() API and the
|
|
8
|
+
Databricks widget notebook is proven to do something, not merely to be
|
|
9
|
+
passed along.
|
|
10
|
+
|
|
11
|
+
DuckDB is a stand-in for Databricks SQL: the validator's generated SQL is
|
|
12
|
+
executed unchanged apart from a handful of dialect renames in
|
|
13
|
+
_DuckDbConnector._execute_to_dataframe (sha2 -> sha256, STRING ->
|
|
14
|
+
VARCHAR, and a conv() shim). Those translations are in the harness only;
|
|
15
|
+
the validator itself is never special-cased.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
21
|
+
|
|
22
|
+
import pandas as pd
|
|
23
|
+
import pytest
|
|
24
|
+
|
|
25
|
+
from table_validator.config.schema import ValidationType
|
|
26
|
+
from table_validator.connectors.base_sql_connector import BaseSqlConnector
|
|
27
|
+
from table_validator.models import CatalogValidationRequest
|
|
28
|
+
from table_validator.notebook import ValidationResult, _parse_source_target
|
|
29
|
+
from table_validator.validators.catalog_validator import CatalogValidator
|
|
30
|
+
|
|
31
|
+
duckdb = pytest.importorskip("duckdb", reason="real-engine tests need duckdb")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _conv(value: str, from_base: int, to_base: int) -> int:
|
|
35
|
+
"""Databricks' conv(str, from_base, to_base), which DuckDB lacks. The
|
|
36
|
+
validator uses it to turn a hex hash slice into a number for
|
|
37
|
+
fingerprinting."""
|
|
38
|
+
try:
|
|
39
|
+
return int(str(value), int(from_base))
|
|
40
|
+
except (TypeError, ValueError):
|
|
41
|
+
return 0
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@pytest.fixture
|
|
45
|
+
def con():
|
|
46
|
+
"""A real database with real rows.
|
|
47
|
+
|
|
48
|
+
orders - identical on both sides EXCEPT one changed amount (id=3)
|
|
49
|
+
customers- identical on both sides
|
|
50
|
+
products - identical on both sides
|
|
51
|
+
"""
|
|
52
|
+
connection = duckdb.connect()
|
|
53
|
+
connection.create_function("conv", _conv, ["VARCHAR", "INTEGER", "INTEGER"], "BIGINT")
|
|
54
|
+
connection.execute("CREATE SCHEMA bronze")
|
|
55
|
+
connection.execute("CREATE SCHEMA silver")
|
|
56
|
+
|
|
57
|
+
connection.execute(
|
|
58
|
+
"""CREATE TABLE bronze.orders AS SELECT * FROM (VALUES
|
|
59
|
+
(1,'alice',100,'active'), (2,'bob',200,'active'),
|
|
60
|
+
(3,'carol',300,'inactive'), (4,'dan',400,'active')
|
|
61
|
+
) t(id, name, amount, status)"""
|
|
62
|
+
)
|
|
63
|
+
connection.execute(
|
|
64
|
+
"""CREATE TABLE silver.orders AS SELECT * FROM (VALUES
|
|
65
|
+
(1,'alice',100,'active'), (2,'bob',200,'active'),
|
|
66
|
+
(3,'carol',999,'inactive'), (4,'dan',400,'active')
|
|
67
|
+
) t(id, name, amount, status)"""
|
|
68
|
+
)
|
|
69
|
+
for schema in ("bronze", "silver"):
|
|
70
|
+
connection.execute(
|
|
71
|
+
f"CREATE TABLE {schema}.customers AS SELECT * FROM "
|
|
72
|
+
"(VALUES (1,'x'), (2,'y')) t(id, label)"
|
|
73
|
+
)
|
|
74
|
+
connection.execute(
|
|
75
|
+
f"CREATE TABLE {schema}.products AS SELECT 1 AS id, 'p' AS sku"
|
|
76
|
+
)
|
|
77
|
+
return connection
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class _DuckDbConnector(BaseSqlConnector):
|
|
81
|
+
"""Executes the validator's real SQL against DuckDB, recording every
|
|
82
|
+
query so tests can assert on what was actually asked of the engine."""
|
|
83
|
+
|
|
84
|
+
def __init__(self, con):
|
|
85
|
+
super().__init__()
|
|
86
|
+
self._con = con
|
|
87
|
+
self.queries: List[str] = []
|
|
88
|
+
|
|
89
|
+
def connect(self) -> None: ...
|
|
90
|
+
def disconnect(self) -> None: ...
|
|
91
|
+
def test_connection(self) -> bool: return True
|
|
92
|
+
|
|
93
|
+
def _qualify(self, catalog: str, schema: str, table: str) -> str:
|
|
94
|
+
# DuckDB has no catalog layer in this harness.
|
|
95
|
+
return f'"{schema}"."{table}"'
|
|
96
|
+
|
|
97
|
+
def _execute_to_dataframe(self, query: str) -> pd.DataFrame:
|
|
98
|
+
self.queries.append(query)
|
|
99
|
+
translated = (
|
|
100
|
+
query.replace("`", '"')
|
|
101
|
+
.replace("sha2(", "sha256(")
|
|
102
|
+
.replace(", 256)", ")")
|
|
103
|
+
.replace("AS STRING", "AS VARCHAR")
|
|
104
|
+
)
|
|
105
|
+
return self._con.execute(translated).df()
|
|
106
|
+
|
|
107
|
+
def get_schemas(self, catalog: str) -> List[str]:
|
|
108
|
+
return ["bronze", "silver"]
|
|
109
|
+
|
|
110
|
+
def get_tables(self, catalog: str, schema: str) -> List[str]:
|
|
111
|
+
rows = self._con.execute(
|
|
112
|
+
"SELECT table_name FROM information_schema.tables WHERE table_schema=?",
|
|
113
|
+
[schema],
|
|
114
|
+
).fetchall()
|
|
115
|
+
return sorted(r[0] for r in rows)
|
|
116
|
+
|
|
117
|
+
def get_table_schema(self, catalog: str, schema: str, table: str) -> pd.DataFrame:
|
|
118
|
+
rows = self._con.execute(
|
|
119
|
+
"SELECT column_name, data_type, is_nullable, ordinal_position "
|
|
120
|
+
"FROM information_schema.columns WHERE table_schema=? AND table_name=? "
|
|
121
|
+
"ORDER BY ordinal_position",
|
|
122
|
+
[schema, table],
|
|
123
|
+
).fetchall()
|
|
124
|
+
return pd.DataFrame(
|
|
125
|
+
[
|
|
126
|
+
{
|
|
127
|
+
"column_name": r[0],
|
|
128
|
+
"data_type": str(r[1]).lower(),
|
|
129
|
+
"is_nullable": r[2] == "YES",
|
|
130
|
+
"ordinal_position": r[3],
|
|
131
|
+
}
|
|
132
|
+
for r in rows
|
|
133
|
+
]
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
def catalog_exists(self, catalog: str) -> bool:
|
|
137
|
+
return True
|
|
138
|
+
|
|
139
|
+
def is_min_max_eligible(self, data_type: str) -> bool:
|
|
140
|
+
return any(k in str(data_type).lower() for k in ("int", "decimal", "double", "float"))
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _run(con, source: str, **kwargs) -> Tuple[ValidationResult, _DuckDbConnector]:
|
|
144
|
+
"""Run the real validator the same way validate_tables() does."""
|
|
145
|
+
target = kwargs.pop("target")
|
|
146
|
+
src_catalog, src_schema, src_table = _parse_source_target(source, "source")
|
|
147
|
+
tgt_catalog, tgt_schema, tgt_table = _parse_source_target(target, "target")
|
|
148
|
+
|
|
149
|
+
connector = _DuckDbConnector(con)
|
|
150
|
+
if any(kwargs.get(k) for k in ("row_filter", "source_row_filter", "target_row_filter")):
|
|
151
|
+
connector.set_row_filters(
|
|
152
|
+
common=kwargs.get("row_filter"),
|
|
153
|
+
source=kwargs.get("source_row_filter"),
|
|
154
|
+
target=kwargs.get("target_row_filter"),
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
primary_key = kwargs.get("primary_key")
|
|
158
|
+
request_kwargs: Dict[str, Any] = dict(
|
|
159
|
+
source_catalog=src_catalog,
|
|
160
|
+
target_catalog=tgt_catalog,
|
|
161
|
+
schemas=[src_schema],
|
|
162
|
+
schema_map={src_schema: tgt_schema} if src_schema != tgt_schema else {},
|
|
163
|
+
tables=[src_table] if src_table else None,
|
|
164
|
+
table_map={},
|
|
165
|
+
primary_keys=(
|
|
166
|
+
{tgt_table: primary_key, f"{tgt_schema}.{tgt_table}": primary_key}
|
|
167
|
+
if primary_key and tgt_table
|
|
168
|
+
else {}
|
|
169
|
+
),
|
|
170
|
+
ignore_columns=kwargs.get("ignore_columns") or [],
|
|
171
|
+
only_columns=kwargs.get("only_columns"),
|
|
172
|
+
column_map=kwargs.get("column_map") or {},
|
|
173
|
+
ignore_datatype_columns=kwargs.get("ignore_datatype_columns") or [],
|
|
174
|
+
)
|
|
175
|
+
if kwargs.get("enabled_validations") is not None:
|
|
176
|
+
request_kwargs["enabled_validations"] = kwargs["enabled_validations"]
|
|
177
|
+
|
|
178
|
+
response = CatalogValidator(connector).compare_catalogs(
|
|
179
|
+
CatalogValidationRequest(**request_kwargs)
|
|
180
|
+
)
|
|
181
|
+
return ValidationResult(response), connector
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _tables_in(result: ValidationResult) -> List[str]:
|
|
185
|
+
return sorted(t.table for s in result.response.schemas for t in s.tables)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
# ---------------------------------------------------------------------------
|
|
189
|
+
# Source/Target Table selection
|
|
190
|
+
# ---------------------------------------------------------------------------
|
|
191
|
+
def test_single_table_selection_compares_only_that_table(con):
|
|
192
|
+
result, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders")
|
|
193
|
+
assert _tables_in(result) == ["orders"]
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def test_a_real_data_difference_is_actually_detected(con):
|
|
197
|
+
"""silver.orders has a changed amount for id=3 - the run must FAIL."""
|
|
198
|
+
result, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders")
|
|
199
|
+
assert result.response.status.value == "FAIL"
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def test_identical_tables_pass(con):
|
|
203
|
+
result, _ = _run(con, "cat.bronze.customers", target="cat.silver.customers")
|
|
204
|
+
assert result.response.status.value == "PASS"
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
# ---------------------------------------------------------------------------
|
|
208
|
+
# Schema-wide sweep - the "(all tables in schema)" widget option
|
|
209
|
+
# ---------------------------------------------------------------------------
|
|
210
|
+
def test_sweep_validates_every_table_not_just_the_first(con):
|
|
211
|
+
"""Reported issue: a sweep appeared to validate only the first table."""
|
|
212
|
+
result, _ = _run(con, "cat.bronze", target="cat.silver")
|
|
213
|
+
assert _tables_in(result) == ["customers", "orders", "products"]
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def test_sweep_still_flags_the_one_bad_table(con):
|
|
217
|
+
result, _ = _run(con, "cat.bronze", target="cat.silver")
|
|
218
|
+
assert result.response.status.value == "FAIL"
|
|
219
|
+
|
|
220
|
+
per_table = {t.table: t.status.value for s in result.response.schemas for t in s.tables}
|
|
221
|
+
assert per_table["orders"] == "FAIL"
|
|
222
|
+
assert per_table["customers"] == "PASS"
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
# ---------------------------------------------------------------------------
|
|
226
|
+
# Row filter widgets
|
|
227
|
+
# ---------------------------------------------------------------------------
|
|
228
|
+
def test_row_filter_reaches_the_real_sql(con):
|
|
229
|
+
_, connector = _run(con, "cat.bronze.orders", target="cat.silver.orders", row_filter="id < 3")
|
|
230
|
+
assert any("WHERE (id < 3)" in q for q in connector.queries)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def test_row_filter_excluding_the_bad_row_turns_fail_into_pass(con):
|
|
234
|
+
"""The strongest proof the filter is real: the same two tables give a
|
|
235
|
+
different verdict purely because of the filter."""
|
|
236
|
+
unfiltered, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders")
|
|
237
|
+
filtered, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders", row_filter="id < 3")
|
|
238
|
+
|
|
239
|
+
assert unfiltered.response.status.value == "FAIL"
|
|
240
|
+
assert filtered.response.status.value == "PASS"
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def test_per_side_row_filters_are_applied_independently(con):
|
|
244
|
+
_, connector = _run(
|
|
245
|
+
con, "cat.bronze.orders", target="cat.silver.orders",
|
|
246
|
+
source_row_filter="id > 1", target_row_filter="id > 2",
|
|
247
|
+
)
|
|
248
|
+
assert any("id > 1" in q for q in connector.queries)
|
|
249
|
+
assert any("id > 2" in q for q in connector.queries)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
# ---------------------------------------------------------------------------
|
|
253
|
+
# Primary key widget
|
|
254
|
+
# ---------------------------------------------------------------------------
|
|
255
|
+
def test_primary_key_is_used_for_row_hashing(con):
|
|
256
|
+
_, connector = _run(
|
|
257
|
+
con, "cat.bronze.orders", target="cat.silver.orders", primary_key=["id"],
|
|
258
|
+
)
|
|
259
|
+
assert any("`id`" in q and "row_hash" in q.lower() for q in connector.queries)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def test_primary_key_pinpoints_the_exact_changed_record(con):
|
|
263
|
+
result, _ = _run(
|
|
264
|
+
con, "cat.bronze.orders", target="cat.silver.orders", primary_key=["id"],
|
|
265
|
+
)
|
|
266
|
+
table = [t for s in result.response.schemas for t in s.tables][0]
|
|
267
|
+
assert table.data.row_hash_mismatch_count == 1
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# ---------------------------------------------------------------------------
|
|
271
|
+
# Column controls
|
|
272
|
+
# ---------------------------------------------------------------------------
|
|
273
|
+
def test_only_columns_never_queries_the_excluded_column(con):
|
|
274
|
+
_, connector = _run(
|
|
275
|
+
con, "cat.bronze.orders", target="cat.silver.orders", only_columns=["id", "name"],
|
|
276
|
+
)
|
|
277
|
+
assert not any("amount" in q for q in connector.queries)
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def test_ignoring_the_differing_column_turns_fail_into_pass(con):
|
|
281
|
+
result, connector = _run(
|
|
282
|
+
con, "cat.bronze.orders", target="cat.silver.orders", ignore_columns=["amount"],
|
|
283
|
+
)
|
|
284
|
+
assert result.response.status.value == "PASS"
|
|
285
|
+
assert not any("amount" in q for q in connector.queries)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
# ---------------------------------------------------------------------------
|
|
289
|
+
# Check-group widgets
|
|
290
|
+
# ---------------------------------------------------------------------------
|
|
291
|
+
def test_disabling_check_groups_does_less_real_work(con):
|
|
292
|
+
"""Turning checks off must genuinely skip queries, not just relabel
|
|
293
|
+
the result."""
|
|
294
|
+
_, few = _run(
|
|
295
|
+
con, "cat.bronze.orders", target="cat.silver.orders",
|
|
296
|
+
enabled_validations={ValidationType.CATALOG, ValidationType.SCHEMA},
|
|
297
|
+
)
|
|
298
|
+
_, many = _run(
|
|
299
|
+
con, "cat.bronze.orders", target="cat.silver.orders",
|
|
300
|
+
enabled_validations={
|
|
301
|
+
ValidationType.CATALOG, ValidationType.SCHEMA,
|
|
302
|
+
ValidationType.COLUMN, ValidationType.ROW,
|
|
303
|
+
},
|
|
304
|
+
)
|
|
305
|
+
assert len(few.queries) < len(many.queries)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def test_catalog_schema_only_reports_skipped(con):
|
|
309
|
+
"""Documented behaviour users hit: this combination produces no
|
|
310
|
+
per-table verdict."""
|
|
311
|
+
result, _ = _run(
|
|
312
|
+
con, "cat.bronze.orders", target="cat.silver.orders",
|
|
313
|
+
enabled_validations={ValidationType.CATALOG, ValidationType.SCHEMA},
|
|
314
|
+
)
|
|
315
|
+
table = [t for s in result.response.schemas for t in s.tables][0]
|
|
316
|
+
assert table.status.value == "SKIPPED"
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
# ---------------------------------------------------------------------------
|
|
320
|
+
# The results a user actually reads
|
|
321
|
+
# ---------------------------------------------------------------------------
|
|
322
|
+
def test_result_sheets_are_arrow_convertible_after_a_real_run(con):
|
|
323
|
+
"""display() in Databricks converts via Arrow - every sheet from a
|
|
324
|
+
real run must survive that."""
|
|
325
|
+
pa = pytest.importorskip("pyarrow")
|
|
326
|
+
|
|
327
|
+
result, _ = _run(con, "cat.bronze", target="cat.silver")
|
|
328
|
+
|
|
329
|
+
for sheet in (
|
|
330
|
+
"table_validation", "column_validation", "data_mismatches",
|
|
331
|
+
"row_hash_mismatches", "mismatch_categories", "suggestions",
|
|
332
|
+
):
|
|
333
|
+
pa.Table.from_pandas(getattr(result, sheet).to_dataframe())
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/azure_connector.py
RENAMED
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/base_sql_connector.py
RENAMED
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/databricks_connector.py
RENAMED
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/spark_connector.py
RENAMED
|
File without changes
|
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/engine/comparison_engine.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/blob_discovery.py
RENAMED
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/catalog_validator.py
RENAMED
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/mismatch_classifier.py
RENAMED
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/row_validator.py
RENAMED
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_validator-0.1.22 → table_validator-0.1.23}/tests/test_package_import_without_pyspark.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|