table-validator 0.1.22__tar.gz → 0.1.23__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. {table_validator-0.1.22/table_validator.egg-info → table_validator-0.1.23}/PKG-INFO +2 -1
  2. {table_validator-0.1.22 → table_validator-0.1.23}/pyproject.toml +4 -0
  3. {table_validator-0.1.22 → table_validator-0.1.23/table_validator.egg-info}/PKG-INFO +2 -1
  4. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/SOURCES.txt +1 -0
  5. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/requires.txt +1 -0
  6. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/scm_file_list.json +1 -0
  7. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/scm_version.json +2 -2
  8. table_validator-0.1.23/tests/test_end_to_end_real_engine.py +333 -0
  9. {table_validator-0.1.22 → table_validator-0.1.23}/LICENSE +0 -0
  10. {table_validator-0.1.22 → table_validator-0.1.23}/README.md +0 -0
  11. {table_validator-0.1.22 → table_validator-0.1.23}/setup.cfg +0 -0
  12. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/__init__.py +0 -0
  13. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/auth/__init__.py +0 -0
  14. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/auth/azure_auth.py +0 -0
  15. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/auth/databricks_auth.py +0 -0
  16. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/__init__.py +0 -0
  17. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/main.py +0 -0
  18. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/partition_prompt.py +0 -0
  19. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/summary_table.py +0 -0
  20. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/cli/wizard.py +0 -0
  21. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/config/__init__.py +0 -0
  22. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/config/manager.py +0 -0
  23. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/config/schema.py +0 -0
  24. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/__init__.py +0 -0
  25. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/azure_connector.py +0 -0
  26. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/base_sql_connector.py +0 -0
  27. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/databricks_connector.py +0 -0
  28. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/connectors/spark_connector.py +0 -0
  29. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/engine/__init__.py +0 -0
  30. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/engine/comparison_engine.py +0 -0
  31. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/models.py +0 -0
  32. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/notebook.py +0 -0
  33. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/reports/__init__.py +0 -0
  34. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/reports/excel_report.py +0 -0
  35. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/__init__.py +0 -0
  36. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/blob_discovery.py +0 -0
  37. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/catalog_validator.py +0 -0
  38. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/mismatch_classifier.py +0 -0
  39. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator/validators/row_validator.py +0 -0
  40. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/dependency_links.txt +0 -0
  41. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/entry_points.txt +0 -0
  42. {table_validator-0.1.22 → table_validator-0.1.23}/table_validator.egg-info/top_level.txt +0 -0
  43. {table_validator-0.1.22 → table_validator-0.1.23}/tests/__init__.py +0 -0
  44. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_base_sql_connector.py +0 -0
  45. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_blob_discovery.py +0 -0
  46. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_catalog_validator.py +0 -0
  47. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_cli.py +0 -0
  48. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_databricks_connector.py +0 -0
  49. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_excel_report.py +0 -0
  50. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_mismatch_classifier.py +0 -0
  51. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_notebook.py +0 -0
  52. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_package_import_without_pyspark.py +0 -0
  53. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_partition_prompt.py +0 -0
  54. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_report_command.py +0 -0
  55. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_row_validator.py +0 -0
  56. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_spark_connector.py +0 -0
  57. {table_validator-0.1.22 → table_validator-0.1.23}/tests/test_wizard.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.22
3
+ Version: 0.1.23
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -24,6 +24,7 @@ Provides-Extra: dev
24
24
  Requires-Dist: pytest>=7.0; extra == "dev"
25
25
  Requires-Dist: ruff>=0.4; extra == "dev"
26
26
  Requires-Dist: black>=24.0; extra == "dev"
27
+ Requires-Dist: duckdb>=1.0; extra == "dev"
27
28
  Provides-Extra: spark
28
29
  Requires-Dist: pyspark>=3.4; extra == "spark"
29
30
  Dynamic: license-file
@@ -31,6 +31,10 @@ dev = [
31
31
  "pytest>=7.0",
32
32
  "ruff>=0.4",
33
33
  "black>=24.0",
34
+ # tests/test_end_to_end_real_engine.py runs the validator against a
35
+ # real SQL engine rather than mocked query results, so that every
36
+ # user-facing option is proven to actually change what gets validated.
37
+ "duckdb>=1.0",
34
38
  ]
35
39
  # Only needed to construct SparkConnector/call validate_tables() OUTSIDE a
36
40
  # real Databricks notebook (e.g. local dev/testing) - inside an actual
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: table-validator
3
- Version: 0.1.22
3
+ Version: 0.1.23
4
4
  Summary: CLI tool for validating data migrations between Azure (Blob Storage / SQL Database) and Databricks Delta Lake catalogs
5
5
  License: MIT
6
6
  Requires-Python: >=3.9
@@ -24,6 +24,7 @@ Provides-Extra: dev
24
24
  Requires-Dist: pytest>=7.0; extra == "dev"
25
25
  Requires-Dist: ruff>=0.4; extra == "dev"
26
26
  Requires-Dist: black>=24.0; extra == "dev"
27
+ Requires-Dist: duckdb>=1.0; extra == "dev"
27
28
  Provides-Extra: spark
28
29
  Requires-Dist: pyspark>=3.4; extra == "spark"
29
30
  Dynamic: license-file
@@ -43,6 +43,7 @@ tests/test_blob_discovery.py
43
43
  tests/test_catalog_validator.py
44
44
  tests/test_cli.py
45
45
  tests/test_databricks_connector.py
46
+ tests/test_end_to_end_real_engine.py
46
47
  tests/test_excel_report.py
47
48
  tests/test_mismatch_classifier.py
48
49
  tests/test_notebook.py
@@ -17,6 +17,7 @@ python-dotenv>=1.0
17
17
  pytest>=7.0
18
18
  ruff>=0.4
19
19
  black>=24.0
20
+ duckdb>=1.0
20
21
 
21
22
  [spark]
22
23
  pyspark>=3.4
@@ -37,6 +37,7 @@
37
37
  "tests/test_catalog_validator.py",
38
38
  "tests/test_cli.py",
39
39
  "tests/test_databricks_connector.py",
40
+ "tests/test_end_to_end_real_engine.py",
40
41
  "tests/test_excel_report.py",
41
42
  "tests/test_mismatch_classifier.py",
42
43
  "tests/test_notebook.py",
@@ -1,7 +1,7 @@
1
1
  {
2
- "tag": "0.1.22",
2
+ "tag": "0.1.23",
3
3
  "distance": 0,
4
- "node": "g9bb35af8f8b4c9edbf94bf61c68df2aed7fbfe08",
4
+ "node": "ga62949fa884aa38b3e70751533b67c62420dacb7",
5
5
  "dirty": false,
6
6
  "branch": "HEAD",
7
7
  "node_date": "2026-09-10"
@@ -0,0 +1,333 @@
1
+ """End-to-end validation against a REAL SQL engine, not mocked results.
2
+
3
+ Every other test file mocks the connector's query results, which proves
4
+ the validator's own logic but not that a user-facing option actually
5
+ changes what gets validated. These tests create real tables in DuckDB,
6
+ run the real validator against them, and assert on real verdicts - so
7
+ each option offered by the CLI wizard, the validate_tables() API and the
8
+ Databricks widget notebook is proven to do something, not merely to be
9
+ passed along.
10
+
11
+ DuckDB is a stand-in for Databricks SQL: the validator's generated SQL is
12
+ executed unchanged apart from a handful of dialect renames in
13
+ _DuckDbConnector._execute_to_dataframe (sha2 -> sha256, STRING ->
14
+ VARCHAR, and a conv() shim). Those translations are in the harness only;
15
+ the validator itself is never special-cased.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from typing import Any, Dict, List, Optional, Tuple
21
+
22
+ import pandas as pd
23
+ import pytest
24
+
25
+ from table_validator.config.schema import ValidationType
26
+ from table_validator.connectors.base_sql_connector import BaseSqlConnector
27
+ from table_validator.models import CatalogValidationRequest
28
+ from table_validator.notebook import ValidationResult, _parse_source_target
29
+ from table_validator.validators.catalog_validator import CatalogValidator
30
+
31
+ duckdb = pytest.importorskip("duckdb", reason="real-engine tests need duckdb")
32
+
33
+
34
+ def _conv(value: str, from_base: int, to_base: int) -> int:
35
+ """Databricks' conv(str, from_base, to_base), which DuckDB lacks. The
36
+ validator uses it to turn a hex hash slice into a number for
37
+ fingerprinting."""
38
+ try:
39
+ return int(str(value), int(from_base))
40
+ except (TypeError, ValueError):
41
+ return 0
42
+
43
+
44
+ @pytest.fixture
45
+ def con():
46
+ """A real database with real rows.
47
+
48
+ orders - identical on both sides EXCEPT one changed amount (id=3)
49
+ customers- identical on both sides
50
+ products - identical on both sides
51
+ """
52
+ connection = duckdb.connect()
53
+ connection.create_function("conv", _conv, ["VARCHAR", "INTEGER", "INTEGER"], "BIGINT")
54
+ connection.execute("CREATE SCHEMA bronze")
55
+ connection.execute("CREATE SCHEMA silver")
56
+
57
+ connection.execute(
58
+ """CREATE TABLE bronze.orders AS SELECT * FROM (VALUES
59
+ (1,'alice',100,'active'), (2,'bob',200,'active'),
60
+ (3,'carol',300,'inactive'), (4,'dan',400,'active')
61
+ ) t(id, name, amount, status)"""
62
+ )
63
+ connection.execute(
64
+ """CREATE TABLE silver.orders AS SELECT * FROM (VALUES
65
+ (1,'alice',100,'active'), (2,'bob',200,'active'),
66
+ (3,'carol',999,'inactive'), (4,'dan',400,'active')
67
+ ) t(id, name, amount, status)"""
68
+ )
69
+ for schema in ("bronze", "silver"):
70
+ connection.execute(
71
+ f"CREATE TABLE {schema}.customers AS SELECT * FROM "
72
+ "(VALUES (1,'x'), (2,'y')) t(id, label)"
73
+ )
74
+ connection.execute(
75
+ f"CREATE TABLE {schema}.products AS SELECT 1 AS id, 'p' AS sku"
76
+ )
77
+ return connection
78
+
79
+
80
+ class _DuckDbConnector(BaseSqlConnector):
81
+ """Executes the validator's real SQL against DuckDB, recording every
82
+ query so tests can assert on what was actually asked of the engine."""
83
+
84
+ def __init__(self, con):
85
+ super().__init__()
86
+ self._con = con
87
+ self.queries: List[str] = []
88
+
89
+ def connect(self) -> None: ...
90
+ def disconnect(self) -> None: ...
91
+ def test_connection(self) -> bool: return True
92
+
93
+ def _qualify(self, catalog: str, schema: str, table: str) -> str:
94
+ # DuckDB has no catalog layer in this harness.
95
+ return f'"{schema}"."{table}"'
96
+
97
+ def _execute_to_dataframe(self, query: str) -> pd.DataFrame:
98
+ self.queries.append(query)
99
+ translated = (
100
+ query.replace("`", '"')
101
+ .replace("sha2(", "sha256(")
102
+ .replace(", 256)", ")")
103
+ .replace("AS STRING", "AS VARCHAR")
104
+ )
105
+ return self._con.execute(translated).df()
106
+
107
+ def get_schemas(self, catalog: str) -> List[str]:
108
+ return ["bronze", "silver"]
109
+
110
+ def get_tables(self, catalog: str, schema: str) -> List[str]:
111
+ rows = self._con.execute(
112
+ "SELECT table_name FROM information_schema.tables WHERE table_schema=?",
113
+ [schema],
114
+ ).fetchall()
115
+ return sorted(r[0] for r in rows)
116
+
117
+ def get_table_schema(self, catalog: str, schema: str, table: str) -> pd.DataFrame:
118
+ rows = self._con.execute(
119
+ "SELECT column_name, data_type, is_nullable, ordinal_position "
120
+ "FROM information_schema.columns WHERE table_schema=? AND table_name=? "
121
+ "ORDER BY ordinal_position",
122
+ [schema, table],
123
+ ).fetchall()
124
+ return pd.DataFrame(
125
+ [
126
+ {
127
+ "column_name": r[0],
128
+ "data_type": str(r[1]).lower(),
129
+ "is_nullable": r[2] == "YES",
130
+ "ordinal_position": r[3],
131
+ }
132
+ for r in rows
133
+ ]
134
+ )
135
+
136
+ def catalog_exists(self, catalog: str) -> bool:
137
+ return True
138
+
139
+ def is_min_max_eligible(self, data_type: str) -> bool:
140
+ return any(k in str(data_type).lower() for k in ("int", "decimal", "double", "float"))
141
+
142
+
143
+ def _run(con, source: str, **kwargs) -> Tuple[ValidationResult, _DuckDbConnector]:
144
+ """Run the real validator the same way validate_tables() does."""
145
+ target = kwargs.pop("target")
146
+ src_catalog, src_schema, src_table = _parse_source_target(source, "source")
147
+ tgt_catalog, tgt_schema, tgt_table = _parse_source_target(target, "target")
148
+
149
+ connector = _DuckDbConnector(con)
150
+ if any(kwargs.get(k) for k in ("row_filter", "source_row_filter", "target_row_filter")):
151
+ connector.set_row_filters(
152
+ common=kwargs.get("row_filter"),
153
+ source=kwargs.get("source_row_filter"),
154
+ target=kwargs.get("target_row_filter"),
155
+ )
156
+
157
+ primary_key = kwargs.get("primary_key")
158
+ request_kwargs: Dict[str, Any] = dict(
159
+ source_catalog=src_catalog,
160
+ target_catalog=tgt_catalog,
161
+ schemas=[src_schema],
162
+ schema_map={src_schema: tgt_schema} if src_schema != tgt_schema else {},
163
+ tables=[src_table] if src_table else None,
164
+ table_map={},
165
+ primary_keys=(
166
+ {tgt_table: primary_key, f"{tgt_schema}.{tgt_table}": primary_key}
167
+ if primary_key and tgt_table
168
+ else {}
169
+ ),
170
+ ignore_columns=kwargs.get("ignore_columns") or [],
171
+ only_columns=kwargs.get("only_columns"),
172
+ column_map=kwargs.get("column_map") or {},
173
+ ignore_datatype_columns=kwargs.get("ignore_datatype_columns") or [],
174
+ )
175
+ if kwargs.get("enabled_validations") is not None:
176
+ request_kwargs["enabled_validations"] = kwargs["enabled_validations"]
177
+
178
+ response = CatalogValidator(connector).compare_catalogs(
179
+ CatalogValidationRequest(**request_kwargs)
180
+ )
181
+ return ValidationResult(response), connector
182
+
183
+
184
+ def _tables_in(result: ValidationResult) -> List[str]:
185
+ return sorted(t.table for s in result.response.schemas for t in s.tables)
186
+
187
+
188
+ # ---------------------------------------------------------------------------
189
+ # Source/Target Table selection
190
+ # ---------------------------------------------------------------------------
191
+ def test_single_table_selection_compares_only_that_table(con):
192
+ result, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders")
193
+ assert _tables_in(result) == ["orders"]
194
+
195
+
196
+ def test_a_real_data_difference_is_actually_detected(con):
197
+ """silver.orders has a changed amount for id=3 - the run must FAIL."""
198
+ result, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders")
199
+ assert result.response.status.value == "FAIL"
200
+
201
+
202
+ def test_identical_tables_pass(con):
203
+ result, _ = _run(con, "cat.bronze.customers", target="cat.silver.customers")
204
+ assert result.response.status.value == "PASS"
205
+
206
+
207
+ # ---------------------------------------------------------------------------
208
+ # Schema-wide sweep - the "(all tables in schema)" widget option
209
+ # ---------------------------------------------------------------------------
210
+ def test_sweep_validates_every_table_not_just_the_first(con):
211
+ """Reported issue: a sweep appeared to validate only the first table."""
212
+ result, _ = _run(con, "cat.bronze", target="cat.silver")
213
+ assert _tables_in(result) == ["customers", "orders", "products"]
214
+
215
+
216
+ def test_sweep_still_flags_the_one_bad_table(con):
217
+ result, _ = _run(con, "cat.bronze", target="cat.silver")
218
+ assert result.response.status.value == "FAIL"
219
+
220
+ per_table = {t.table: t.status.value for s in result.response.schemas for t in s.tables}
221
+ assert per_table["orders"] == "FAIL"
222
+ assert per_table["customers"] == "PASS"
223
+
224
+
225
+ # ---------------------------------------------------------------------------
226
+ # Row filter widgets
227
+ # ---------------------------------------------------------------------------
228
+ def test_row_filter_reaches_the_real_sql(con):
229
+ _, connector = _run(con, "cat.bronze.orders", target="cat.silver.orders", row_filter="id < 3")
230
+ assert any("WHERE (id < 3)" in q for q in connector.queries)
231
+
232
+
233
+ def test_row_filter_excluding_the_bad_row_turns_fail_into_pass(con):
234
+ """The strongest proof the filter is real: the same two tables give a
235
+ different verdict purely because of the filter."""
236
+ unfiltered, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders")
237
+ filtered, _ = _run(con, "cat.bronze.orders", target="cat.silver.orders", row_filter="id < 3")
238
+
239
+ assert unfiltered.response.status.value == "FAIL"
240
+ assert filtered.response.status.value == "PASS"
241
+
242
+
243
+ def test_per_side_row_filters_are_applied_independently(con):
244
+ _, connector = _run(
245
+ con, "cat.bronze.orders", target="cat.silver.orders",
246
+ source_row_filter="id > 1", target_row_filter="id > 2",
247
+ )
248
+ assert any("id > 1" in q for q in connector.queries)
249
+ assert any("id > 2" in q for q in connector.queries)
250
+
251
+
252
+ # ---------------------------------------------------------------------------
253
+ # Primary key widget
254
+ # ---------------------------------------------------------------------------
255
+ def test_primary_key_is_used_for_row_hashing(con):
256
+ _, connector = _run(
257
+ con, "cat.bronze.orders", target="cat.silver.orders", primary_key=["id"],
258
+ )
259
+ assert any("`id`" in q and "row_hash" in q.lower() for q in connector.queries)
260
+
261
+
262
+ def test_primary_key_pinpoints_the_exact_changed_record(con):
263
+ result, _ = _run(
264
+ con, "cat.bronze.orders", target="cat.silver.orders", primary_key=["id"],
265
+ )
266
+ table = [t for s in result.response.schemas for t in s.tables][0]
267
+ assert table.data.row_hash_mismatch_count == 1
268
+
269
+
270
+ # ---------------------------------------------------------------------------
271
+ # Column controls
272
+ # ---------------------------------------------------------------------------
273
+ def test_only_columns_never_queries_the_excluded_column(con):
274
+ _, connector = _run(
275
+ con, "cat.bronze.orders", target="cat.silver.orders", only_columns=["id", "name"],
276
+ )
277
+ assert not any("amount" in q for q in connector.queries)
278
+
279
+
280
+ def test_ignoring_the_differing_column_turns_fail_into_pass(con):
281
+ result, connector = _run(
282
+ con, "cat.bronze.orders", target="cat.silver.orders", ignore_columns=["amount"],
283
+ )
284
+ assert result.response.status.value == "PASS"
285
+ assert not any("amount" in q for q in connector.queries)
286
+
287
+
288
+ # ---------------------------------------------------------------------------
289
+ # Check-group widgets
290
+ # ---------------------------------------------------------------------------
291
+ def test_disabling_check_groups_does_less_real_work(con):
292
+ """Turning checks off must genuinely skip queries, not just relabel
293
+ the result."""
294
+ _, few = _run(
295
+ con, "cat.bronze.orders", target="cat.silver.orders",
296
+ enabled_validations={ValidationType.CATALOG, ValidationType.SCHEMA},
297
+ )
298
+ _, many = _run(
299
+ con, "cat.bronze.orders", target="cat.silver.orders",
300
+ enabled_validations={
301
+ ValidationType.CATALOG, ValidationType.SCHEMA,
302
+ ValidationType.COLUMN, ValidationType.ROW,
303
+ },
304
+ )
305
+ assert len(few.queries) < len(many.queries)
306
+
307
+
308
+ def test_catalog_schema_only_reports_skipped(con):
309
+ """Documented behaviour users hit: this combination produces no
310
+ per-table verdict."""
311
+ result, _ = _run(
312
+ con, "cat.bronze.orders", target="cat.silver.orders",
313
+ enabled_validations={ValidationType.CATALOG, ValidationType.SCHEMA},
314
+ )
315
+ table = [t for s in result.response.schemas for t in s.tables][0]
316
+ assert table.status.value == "SKIPPED"
317
+
318
+
319
+ # ---------------------------------------------------------------------------
320
+ # The results a user actually reads
321
+ # ---------------------------------------------------------------------------
322
+ def test_result_sheets_are_arrow_convertible_after_a_real_run(con):
323
+ """display() in Databricks converts via Arrow - every sheet from a
324
+ real run must survive that."""
325
+ pa = pytest.importorskip("pyarrow")
326
+
327
+ result, _ = _run(con, "cat.bronze", target="cat.silver")
328
+
329
+ for sheet in (
330
+ "table_validation", "column_validation", "data_mismatches",
331
+ "row_hash_mismatches", "mismatch_categories", "suggestions",
332
+ ):
333
+ pa.Table.from_pandas(getattr(result, sheet).to_dataframe())