tablevalidator-databricks 0.1.5__tar.gz → 0.1.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/PKG-INFO +1 -1
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks/cli/main.py +46 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks/generator/generator.py +3 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks/templates/widget_notebook.py.tmpl +51 -34
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/PKG-INFO +1 -1
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/scm_version.json +2 -2
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tests/test_cli.py +33 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tests/test_generator.py +15 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/README.md +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/pyproject.toml +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/setup.cfg +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks/__init__.py +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks/cli/__init__.py +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks/generator/__init__.py +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/SOURCES.txt +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/dependency_links.txt +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/entry_points.txt +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/requires.txt +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/scm_file_list.json +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tablevalidator_databricks.egg-info/top_level.txt +0 -0
- {tablevalidator_databricks-0.1.5 → tablevalidator_databricks-0.1.6}/tests/test_generated_notebook_execution.py +0 -0
|
@@ -1,9 +1,11 @@
|
|
|
1
1
|
"""CLI entry point: `tablevalidator-databricks` console script."""
|
|
2
2
|
|
|
3
3
|
from pathlib import Path
|
|
4
|
+
from typing import Optional
|
|
4
5
|
|
|
5
6
|
import typer
|
|
6
7
|
|
|
8
|
+
from tablevalidator_databricks import __version__
|
|
7
9
|
from tablevalidator_databricks.generator.generator import (
|
|
8
10
|
VALID_MODES,
|
|
9
11
|
_min_core_version,
|
|
@@ -22,6 +24,25 @@ app = typer.Typer(
|
|
|
22
24
|
DEFAULT_OUTPUT_PATH = Path("TableValidator.py")
|
|
23
25
|
|
|
24
26
|
|
|
27
|
+
def _version_callback(value: bool) -> None:
|
|
28
|
+
if value:
|
|
29
|
+
typer.echo(f"tablevalidator-databricks {__version__}")
|
|
30
|
+
raise typer.Exit()
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@app.callback()
|
|
34
|
+
def _main(
|
|
35
|
+
version: Optional[bool] = typer.Option(
|
|
36
|
+
None,
|
|
37
|
+
"--version",
|
|
38
|
+
callback=_version_callback,
|
|
39
|
+
is_eager=True,
|
|
40
|
+
help="Show the installed version and exit.",
|
|
41
|
+
),
|
|
42
|
+
) -> None:
|
|
43
|
+
"""Generate a widget-driven Databricks notebook UI for table-validator."""
|
|
44
|
+
|
|
45
|
+
|
|
25
46
|
@app.command()
|
|
26
47
|
def init(
|
|
27
48
|
output: Path = typer.Option(
|
|
@@ -62,10 +83,16 @@ def init(
|
|
|
62
83
|
generate_notebook(output, mode)
|
|
63
84
|
|
|
64
85
|
typer.echo(f"Generated notebook: {output.resolve()}")
|
|
86
|
+
typer.echo(f" generator version: {__version__} (mode: {mode})")
|
|
65
87
|
typer.echo(
|
|
66
88
|
"Import it into Databricks (Workspace -> Import, format 'Source'), "
|
|
67
89
|
"fill in the widgets, and Run All."
|
|
68
90
|
)
|
|
91
|
+
typer.echo(
|
|
92
|
+
"Re-import over the old notebook if you are updating one - widgets are "
|
|
93
|
+
"baked in at generation time, so an existing notebook keeps its old "
|
|
94
|
+
"code until you replace it."
|
|
95
|
+
)
|
|
69
96
|
|
|
70
97
|
|
|
71
98
|
_INFO_TEXT = """
|
|
@@ -164,6 +191,25 @@ something far too old and fails with:
|
|
|
164
191
|
|
|
165
192
|
Re-run 'tablevalidator-databricks init' after upgrading this package -
|
|
166
193
|
an already-generated notebook keeps whatever code it was written with.
|
|
194
|
+
|
|
195
|
+
A WIDGET IS MISSING FROM MY NOTEBOOK
|
|
196
|
+
------------------------------------
|
|
197
|
+
Widgets are baked into the .py file when it is generated. Upgrading this
|
|
198
|
+
package does NOT change a notebook that already exists - you have to
|
|
199
|
+
generate a new one and re-import it.
|
|
200
|
+
|
|
201
|
+
Check what generated the notebook you are looking at: the title cell at
|
|
202
|
+
the top names the version. Compare it against your installed version:
|
|
203
|
+
tablevalidator-databricks --version
|
|
204
|
+
|
|
205
|
+
Then regenerate and re-import:
|
|
206
|
+
pip install --upgrade tablevalidator-databricks
|
|
207
|
+
tablevalidator-databricks init
|
|
208
|
+
|
|
209
|
+
For reference, widgets arrived in these versions:
|
|
210
|
+
0.1.1 "(manual override)" text box beside each dropdown
|
|
211
|
+
0.1.2 "(all tables in schema)" schema-wide sweep option
|
|
212
|
+
0.1.3 Yes/No dropdowns per check group (replacing a multiselect)
|
|
167
213
|
""".strip("\n")
|
|
168
214
|
|
|
169
215
|
|
|
@@ -56,10 +56,13 @@ def generate_notebook(output: Path, mode: str) -> None:
|
|
|
56
56
|
.joinpath("widget_notebook.py.tmpl")
|
|
57
57
|
.read_text(encoding="utf-8")
|
|
58
58
|
)
|
|
59
|
+
from tablevalidator_databricks import __version__
|
|
60
|
+
|
|
59
61
|
rendered = (
|
|
60
62
|
template.replace("{{MODE}}", mode)
|
|
61
63
|
.replace("{{SHOW_OPTIONAL_WIDGETS}}", "True" if mode == "full" else "False")
|
|
62
64
|
.replace("{{MIN_CORE_VERSION}}", _min_core_version())
|
|
65
|
+
.replace("{{GENERATOR_VERSION}}", __version__)
|
|
63
66
|
)
|
|
64
67
|
|
|
65
68
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
@@ -1,12 +1,22 @@
|
|
|
1
1
|
# Databricks notebook source
|
|
2
2
|
# MAGIC %md
|
|
3
3
|
# MAGIC # Table Validator
|
|
4
|
-
# MAGIC Generated by `tablevalidator-databricks
|
|
4
|
+
# MAGIC Generated by `tablevalidator-databricks` **v{{GENERATOR_VERSION}}**
|
|
5
|
+
# MAGIC (`init --mode {{MODE}}`).
|
|
5
6
|
# MAGIC
|
|
6
|
-
# MAGIC
|
|
7
|
-
# MAGIC
|
|
8
|
-
# MAGIC
|
|
9
|
-
# MAGIC `
|
|
7
|
+
# MAGIC Widgets are baked in when this file is generated - upgrading the
|
|
8
|
+
# MAGIC package does NOT change a notebook that already exists. If a widget
|
|
9
|
+
# MAGIC described below is missing here, this notebook was generated by an
|
|
10
|
+
# MAGIC older version: run `pip install --upgrade tablevalidator-databricks`
|
|
11
|
+
# MAGIC then `tablevalidator-databricks init` again, and re-import the result.
|
|
12
|
+
# MAGIC
|
|
13
|
+
# MAGIC **First time here?** Choose **Run All**. Widgets are created by running
|
|
14
|
+
# MAGIC cells, so on a fresh notebook they only appear once the run reaches the
|
|
15
|
+
# MAGIC "Create / refresh widgets" cell - that is normal, not a fault. Once they
|
|
16
|
+
# MAGIC are there, fill them in and run again.
|
|
17
|
+
# MAGIC
|
|
18
|
+
# MAGIC No code needs to be written or edited - every cell below just reads the
|
|
19
|
+
# MAGIC widgets and calls the `table_validator` package's `validate_tables()` API.
|
|
10
20
|
# MAGIC
|
|
11
21
|
# MAGIC **Changing Catalog or Schema after the dropdowns are already built?**
|
|
12
22
|
# MAGIC Databricks widgets have no on-change event, so picking a new Catalog
|
|
@@ -176,9 +186,44 @@ def _build_cascading_widgets(prefix: str, label_prefix: str) -> None:
|
|
|
176
186
|
dbutils.widgets.text(f"{prefix}_table_override", "", f"{label_prefix} Table (manual override)")
|
|
177
187
|
|
|
178
188
|
|
|
189
|
+
def _build_check_widgets() -> None:
|
|
190
|
+
"""The three check-group Yes/No dropdowns, plus (in full mode) the
|
|
191
|
+
optional filter/column text boxes.
|
|
192
|
+
|
|
193
|
+
One Yes/No dropdown per check group, rather than a single multiselect -
|
|
194
|
+
dbutils.widgets.multiselect enforces an internal "selection sequence"
|
|
195
|
+
validation against its own default that was observed to reject even a
|
|
196
|
+
same-length, same-items default on some Databricks runtimes/existing-
|
|
197
|
+
widget states (DefaultValueNotInChoicesList), with no reliable
|
|
198
|
+
workaround. Three independent dropdowns have no such cross-widget
|
|
199
|
+
validation and are the more robust primitive for this."""
|
|
200
|
+
yes_no = ["Yes", "No"]
|
|
201
|
+
# --mode schema means "schema SHAPE only" - that shape (column names,
|
|
202
|
+
# types, nullability) is validated by the COLUMN checks, so Column
|
|
203
|
+
# stays on and only the expensive Row checks (row counts + row-level
|
|
204
|
+
# data) are off by default. Turning Column off too would leave nothing
|
|
205
|
+
# producing a per-table verdict, reporting every table as SKIPPED.
|
|
206
|
+
row_default = "No" if MODE == "schema" else "Yes"
|
|
207
|
+
|
|
208
|
+
dbutils.widgets.dropdown("check_catalog_schema", "Yes", yes_no, "Run Catalog & Schema checks?")
|
|
209
|
+
dbutils.widgets.dropdown("check_column", "Yes", yes_no, "Run Column checks?")
|
|
210
|
+
dbutils.widgets.dropdown("check_row", row_default, yes_no, "Run Row checks?")
|
|
211
|
+
|
|
212
|
+
if SHOW_OPTIONAL_WIDGETS:
|
|
213
|
+
dbutils.widgets.text("primary_key", "", "Primary key column(s) (comma-separated, optional)")
|
|
214
|
+
dbutils.widgets.text("only_columns", "", "Only compare these columns (comma-separated, optional)")
|
|
215
|
+
dbutils.widgets.text("ignore_columns", "", "Skip these columns entirely (comma-separated, optional)")
|
|
216
|
+
dbutils.widgets.text("row_filter", "", "Row filter - SQL WHERE-fragment (optional)")
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
# Every widget is created here, in ONE cell, so they all appear together
|
|
220
|
+
# and in a predictable order - Databricks lays widgets out in creation
|
|
221
|
+
# order, so splitting these across cells made them interleave oddly and
|
|
222
|
+
# made later ones seem to "appear" partway through a Run All.
|
|
179
223
|
catalogs = _list_catalogs()
|
|
180
224
|
_build_cascading_widgets("source", "Source")
|
|
181
225
|
_build_cascading_widgets("target", "Target")
|
|
226
|
+
_build_check_widgets()
|
|
182
227
|
|
|
183
228
|
print(
|
|
184
229
|
"Widgets refreshed for the CURRENT Catalog/Schema selection.\n"
|
|
@@ -188,7 +233,7 @@ print(
|
|
|
188
233
|
|
|
189
234
|
# COMMAND ----------
|
|
190
235
|
|
|
191
|
-
# MAGIC %md ###
|
|
236
|
+
# MAGIC %md ### About the check groups
|
|
192
237
|
# MAGIC "Catalog & Schema" only checks that the catalogs/schemas/tables exist
|
|
193
238
|
# MAGIC and match by name - it produces no per-table PASS/FAIL on its own, so
|
|
194
239
|
# MAGIC running it alone reports every table as **SKIPPED**. That's expected,
|
|
@@ -199,34 +244,6 @@ print(
|
|
|
199
244
|
|
|
200
245
|
# COMMAND ----------
|
|
201
246
|
|
|
202
|
-
# One Yes/No dropdown per check group, instead of a single multiselect -
|
|
203
|
-
# dbutils.widgets.multiselect enforces an internal "selection sequence"
|
|
204
|
-
# validation against its own default that was observed to reject even a
|
|
205
|
-
# same-length, same-items default on some Databricks runtimes/existing-
|
|
206
|
-
# widget states (DefaultValueNotInChoicesList), with no reliable
|
|
207
|
-
# workaround. Three independent dropdowns have no such cross-widget
|
|
208
|
-
# validation and are the more robust primitive for this.
|
|
209
|
-
_YES_NO = ["Yes", "No"]
|
|
210
|
-
_schema_default = "Yes"
|
|
211
|
-
# --mode schema means "schema SHAPE only" - that shape (column names,
|
|
212
|
-
# types, nullability) is validated by the COLUMN checks, so Column stays
|
|
213
|
-
# on and only the expensive Row checks (row counts + row-level data) are
|
|
214
|
-
# off by default. Turning Column off too would leave nothing producing a
|
|
215
|
-
# per-table verdict, reporting every table as SKIPPED.
|
|
216
|
-
_column_default = "Yes"
|
|
217
|
-
_row_default = "No" if MODE == "schema" else "Yes"
|
|
218
|
-
dbutils.widgets.dropdown("check_catalog_schema", _schema_default, _YES_NO, "Run Catalog & Schema checks?")
|
|
219
|
-
dbutils.widgets.dropdown("check_column", _column_default, _YES_NO, "Run Column checks?")
|
|
220
|
-
dbutils.widgets.dropdown("check_row", _row_default, _YES_NO, "Run Row checks?")
|
|
221
|
-
|
|
222
|
-
if SHOW_OPTIONAL_WIDGETS:
|
|
223
|
-
dbutils.widgets.text("only_columns", "", "Only compare these columns (comma-separated, optional)")
|
|
224
|
-
dbutils.widgets.text("ignore_columns", "", "Skip these columns entirely (comma-separated, optional)")
|
|
225
|
-
dbutils.widgets.text("row_filter", "", "Row filter - SQL WHERE-fragment (optional)")
|
|
226
|
-
dbutils.widgets.text("primary_key", "", "Primary key column(s) (comma-separated, optional)")
|
|
227
|
-
|
|
228
|
-
# COMMAND ----------
|
|
229
|
-
|
|
230
247
|
# MAGIC %md ### Run validation
|
|
231
248
|
|
|
232
249
|
# COMMAND ----------
|
|
@@ -88,3 +88,36 @@ def test_info_renders_the_real_version_floor() -> None:
|
|
|
88
88
|
|
|
89
89
|
assert "{MIN_CORE_VERSION}" not in result.output
|
|
90
90
|
assert f'table-validator>={_min_core_version()}' in result.output
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_version_flag_reports_installed_version() -> None:
|
|
94
|
+
"""`--version` exists so a user can check what they have installed
|
|
95
|
+
against what generated a notebook - the fastest way to diagnose a
|
|
96
|
+
"my widgets look different" report."""
|
|
97
|
+
from tablevalidator_databricks import __version__
|
|
98
|
+
|
|
99
|
+
result = runner.invoke(app, ["--version"])
|
|
100
|
+
|
|
101
|
+
assert result.exit_code == 0
|
|
102
|
+
assert __version__ in result.output
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def test_init_prints_the_generator_version(tmp_path: Path) -> None:
|
|
106
|
+
"""init names the version it generated with, so the number is visible
|
|
107
|
+
at the moment the file is created, not just inside it."""
|
|
108
|
+
from tablevalidator_databricks import __version__
|
|
109
|
+
|
|
110
|
+
result = runner.invoke(app, ["init", "--output", str(tmp_path / "nb.py")])
|
|
111
|
+
|
|
112
|
+
assert result.exit_code == 0
|
|
113
|
+
assert __version__ in result.output
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def test_info_explains_missing_widgets() -> None:
|
|
117
|
+
"""The "a widget is missing" case is a real, reported confusion -
|
|
118
|
+
info must explain that widgets are baked in at generation time."""
|
|
119
|
+
result = runner.invoke(app, ["info"])
|
|
120
|
+
|
|
121
|
+
assert "WIDGET IS MISSING" in result.output
|
|
122
|
+
assert "--version" in result.output
|
|
123
|
+
assert "0.1.3" in result.output # the version-history table
|
|
@@ -116,3 +116,18 @@ def test_min_core_version_reads_from_package_metadata() -> None:
|
|
|
116
116
|
dependency, so the two can never silently drift apart."""
|
|
117
117
|
assert _min_core_version() != "0.0.0"
|
|
118
118
|
assert re.match(r"^[0-9]+\.[0-9]+\.[0-9]+$", _min_core_version())
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_generated_notebook_stamps_the_generator_version(tmp_path: Path) -> None:
|
|
122
|
+
"""The notebook records which version generated it, so a user seeing
|
|
123
|
+
unexpected widgets can tell at a glance whether their file is stale."""
|
|
124
|
+
from tablevalidator_databricks import __version__
|
|
125
|
+
|
|
126
|
+
output = tmp_path / "TableValidator.py"
|
|
127
|
+
generate_notebook(output, "full")
|
|
128
|
+
content = output.read_text(encoding="utf-8")
|
|
129
|
+
|
|
130
|
+
assert __version__ in content
|
|
131
|
+
# And it explains that regenerating is what picks up new widgets.
|
|
132
|
+
assert "upgrading the" in content.lower()
|
|
133
|
+
assert "tablevalidator-databricks init" in content
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|