dataowl 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. dataowl-0.1.0/.gitignore +10 -0
  2. dataowl-0.1.0/CHANGELOG.md +32 -0
  3. dataowl-0.1.0/LICENSE +21 -0
  4. dataowl-0.1.0/PKG-INFO +158 -0
  5. dataowl-0.1.0/README.md +128 -0
  6. dataowl-0.1.0/pyproject.toml +70 -0
  7. dataowl-0.1.0/src/dataowl/__init__.py +62 -0
  8. dataowl-0.1.0/src/dataowl/collect/__init__.py +40 -0
  9. dataowl-0.1.0/src/dataowl/collect/columns.py +135 -0
  10. dataowl-0.1.0/src/dataowl/collect/counts.py +46 -0
  11. dataowl-0.1.0/src/dataowl/collect/detail.py +80 -0
  12. dataowl-0.1.0/src/dataowl/collect/history.py +1 -0
  13. dataowl-0.1.0/src/dataowl/collect/keys.py +1 -0
  14. dataowl-0.1.0/src/dataowl/collect/properties.py +75 -0
  15. dataowl-0.1.0/src/dataowl/collect/tables.py +80 -0
  16. dataowl-0.1.0/src/dataowl/collect/timestamps.py +1 -0
  17. dataowl-0.1.0/src/dataowl/errors.py +7 -0
  18. dataowl-0.1.0/src/dataowl/identifiers.py +112 -0
  19. dataowl-0.1.0/src/dataowl/model/__init__.py +3 -0
  20. dataowl-0.1.0/src/dataowl/model/analysis.py +1 -0
  21. dataowl-0.1.0/src/dataowl/model/facts.py +75 -0
  22. dataowl-0.1.0/src/dataowl/model/history.py +1 -0
  23. dataowl-0.1.0/src/dataowl/model/overview.py +130 -0
  24. dataowl-0.1.0/src/dataowl/py.typed +0 -0
  25. dataowl-0.1.0/src/dataowl/render/__init__.py +3 -0
  26. dataowl-0.1.0/src/dataowl/render/format.py +44 -0
  27. dataowl-0.1.0/src/dataowl/render/terminal.py +138 -0
  28. dataowl-0.1.0/src/dataowl/runner.py +47 -0
  29. dataowl-0.1.0/tests/conftest.py +113 -0
  30. dataowl-0.1.0/tests/fixtures/describe_detail_table.json +20 -0
  31. dataowl-0.1.0/tests/fixtures/overview_table.txt +22 -0
  32. dataowl-0.1.0/tests/fixtures/overview_unavailable.txt +18 -0
  33. dataowl-0.1.0/tests/fixtures/overview_view.txt +20 -0
  34. dataowl-0.1.0/tests/integration/setup_test_tables.sql +289 -0
  35. dataowl-0.1.0/tests/integration/test_integration.py +208 -0
  36. dataowl-0.1.0/tests/unit/test_collect.py +44 -0
  37. dataowl-0.1.0/tests/unit/test_collect_columns.py +299 -0
  38. dataowl-0.1.0/tests/unit/test_collect_counts.py +102 -0
  39. dataowl-0.1.0/tests/unit/test_collect_detail.py +148 -0
  40. dataowl-0.1.0/tests/unit/test_collect_properties.py +178 -0
  41. dataowl-0.1.0/tests/unit/test_collect_tables.py +144 -0
  42. dataowl-0.1.0/tests/unit/test_facts.py +135 -0
  43. dataowl-0.1.0/tests/unit/test_fake_runner.py +81 -0
  44. dataowl-0.1.0/tests/unit/test_format.py +62 -0
  45. dataowl-0.1.0/tests/unit/test_identifiers.py +167 -0
  46. dataowl-0.1.0/tests/unit/test_import.py +7 -0
  47. dataowl-0.1.0/tests/unit/test_import_boundaries.py +31 -0
  48. dataowl-0.1.0/tests/unit/test_inspect.py +298 -0
  49. dataowl-0.1.0/tests/unit/test_no_judgement.py +35 -0
  50. dataowl-0.1.0/tests/unit/test_read_only.py +65 -0
  51. dataowl-0.1.0/tests/unit/test_runner.py +106 -0
  52. dataowl-0.1.0/tests/unit/test_terminal.py +201 -0
@@ -0,0 +1,10 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ build/
5
+ dist/
6
+ *.egg-info/
7
+ .pytest_cache/
8
+ .ruff_cache/
9
+ .mcp/
10
+ build_plan.md
@@ -0,0 +1,32 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ ## [Unreleased]
6
+
7
+ ## [0.1.0] - 2026-10-08
8
+
9
+ First release: overview facts for a table (`inspect`).
10
+
11
+ ### Added
12
+ - `dataowl.inspect(table, *, spark=None, count_views=False)` returning an `Overview` with
13
+ `to_dict()` and `show()`.
14
+ - Facts from `information_schema.tables`: object type (with the raw `table_type`), format,
15
+ owner, comment and created time.
16
+ - Facts from `DESCRIBE DETAIL`: size, number of files, average file size, last data
17
+ modification, partition and clustering columns. Not run for views.
18
+ - Schema from `information_schema.columns`, and the field count including nested fields
19
+ from the Spark schema.
20
+ - Exact row count with `COUNT(*)`. Skipped for views, materialized views, foreign tables and
21
+ objects whose type cannot be read, unless `count_views=True`.
22
+ - Change Data Feed, log retention and deleted file retention from `SHOW TBLPROPERTIES`. Not
23
+ run for views.
24
+ - Every fact carries its source (`metadata`, `exact`, `derived`). Facts that cannot be
25
+ collected are marked unavailable with a reason; only a missing table raises
26
+ (`TableNotFoundError`).
27
+ - Terminal rendering of the overview.
28
+
29
+ ### Known deviations
30
+ - `information_schema.columns.ordinal_position` is observed 0-based in Databricks, although
31
+ the documentation says it is numbered from 1. `ColumnInfo.position` keeps the raw value;
32
+ the `#` column in the output is a running number from 1.
dataowl-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Alexandro Dronnen
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
dataowl-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,158 @@
1
+ Metadata-Version: 2.5
2
+ Name: dataowl
3
+ Version: 0.1.0
4
+ Summary: Facts about Databricks data products for building dbt staging models.
5
+ Project-URL: Homepage, https://github.com/alexD1990/dataowl
6
+ Project-URL: Repository, https://github.com/alexD1990/dataowl
7
+ Project-URL: Changelog, https://github.com/alexD1990/dataowl/blob/master/CHANGELOG.md
8
+ Project-URL: Issues, https://github.com/alexD1990/dataowl/issues
9
+ Author: Alexandro Dronnen
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: data-engineering,databricks,dbt,delta-lake,unity-catalog
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3 :: Only
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Database
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.10
24
+ Provides-Extra: dev
25
+ Requires-Dist: mypy==2.3.1; extra == 'dev'
26
+ Requires-Dist: pyspark>=3.4; extra == 'dev'
27
+ Requires-Dist: pytest; extra == 'dev'
28
+ Requires-Dist: ruff==0.16.9; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # dataowl
32
+
33
+ dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
34
+ rows, schema and table properties. It is meant for data engineers who build a dbt staging
35
+ model and want the facts without writing exploratory SQL. It reports what exists and never
36
+ gives recommendations.
37
+
38
+ ## Requirements
39
+
40
+ - Unity Catalog.
41
+ - Databricks Runtime 14.3 LTS or above, or serverless compute.
42
+ Verified on serverless (Spark Connect).
43
+ - PySpark is provided by Databricks. dataowl has no other runtime dependencies.
44
+
45
+ ## Installation
46
+
47
+ In a Databricks notebook:
48
+
49
+ ```python
50
+ %pip install dataowl==0.1.0
51
+ ```
52
+
53
+ ## Usage
54
+
55
+ ```python
56
+ import dataowl
57
+
58
+ dataowl.inspect("catalog.schema.table").show()
59
+ ```
60
+
61
+ As a JSON-serializable dict, where every fact has a value, a source and a reason when it
62
+ is unavailable:
63
+
64
+ ```python
65
+ overview = dataowl.inspect("catalog.schema.table")
66
+ overview.to_dict()
67
+ ```
68
+
69
+ Row counts are skipped for views, materialized views and foreign tables unless you ask for
70
+ them:
71
+
72
+ ```python
73
+ dataowl.inspect("catalog.schema.some_view", count_views=True).show()
74
+ ```
75
+
76
+ Names with characters other than letters, digits and underscore are quoted with backticks:
77
+ `` dataowl.inspect("`my-catalog`.schema.table") ``.
78
+
79
+ ## Example output
80
+
81
+ Output for Databricks' public sample data (`samples.nyctaxi.trips`):
82
+
83
+ ```text
84
+ samples.nyctaxi.trips (MANAGED, DELTA)
85
+
86
+ Size: 354.2 KB
87
+ Files: 1 (avg 354.2 KB)
88
+ Rows: 21 932
89
+ Columns: 6 (6 incl. nested)
90
+ Partitioned by: –
91
+ Clustered by: –
92
+ Change Data Feed: true
93
+ Log retention: not set (default)
94
+ Deleted file retention: not set (default)
95
+ Created: 2025-09-30 11:28
96
+ Last modified: 2026-09-14 15:07
97
+ Owner: System user
98
+ Comment: –
99
+
100
+ SCHEMA
101
+ # name type nullable comment
102
+ 1 tpep_pickup_datetime timestamp yes
103
+ 2 tpep_dropoff_datetime timestamp yes
104
+ 3 trip_distance double yes
105
+ 4 fare_amount double yes
106
+ 5 pickup_zip int yes
107
+ 6 dropoff_zip int yes
108
+ ```
109
+
110
+ ## Principles
111
+
112
+ - **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
113
+ analyzed.
114
+ - **Facts only.** No recommendations or assessments.
115
+ - **No row values leave Spark.** Only aggregates and metadata are collected.
116
+ - **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
117
+ object type make the affected facts `n/a (<reason>)`; the rest of the report is still
118
+ produced. Only a table that does not exist or is not accessible raises
119
+ `TableNotFoundError`.
120
+
121
+ ## Facts per object type
122
+
123
+ | Facts | Source | MANAGED, EXTERNAL | STREAMING_TABLE, MATERIALIZED_VIEW | FOREIGN | VIEW | UNKNOWN or n/a type |
124
+ |---|---|---|---|---|---|---|
125
+ | Object type, format, owner, comment, created | `information_schema.tables` | yes | yes | yes | yes | yes (n/a if the query fails) |
126
+ | Size, files, avg file size, last modified, partitioning, clustering | `DESCRIBE DETAIL` | yes | attempted¹ | attempted¹ | no | attempted¹ |
127
+ | Columns and schema | `information_schema.columns` | yes | yes | yes | yes | yes |
128
+ | Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
129
+ | Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
130
+ | Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
131
+
132
+ ¹ The query runs. If Databricks does not support it for the object, the facts are shown as
133
+ `n/a (<reason>)`.
134
+
135
+ "UNKNOWN" is a `table_type` dataowl does not recognize, such as `MANAGED_SHALLOW_CLONE`. The
136
+ original value is shown in the header. "n/a type" means the object type could not be read.
137
+
138
+ ## Fields
139
+
140
+ - **Last modified** is the last data change, from `DESCRIBE DETAIL`. **Created** is from
141
+ `information_schema.tables`.
142
+ - **Rows** is an exact `COUNT(*)`. On large tables this costs compute.
143
+ - **Files (avg ...)** is the size divided by the number of files.
144
+ - **Columns (... incl. nested)** counts every field, including fields inside structs, array
145
+ elements and map keys and values.
146
+ - **not set (default)** means the table property is not set, so Databricks uses its default.
147
+ dataowl does not report what the default is.
148
+ - **#** in the schema table is a running number from 1 in column order.
149
+
150
+ ## Known deviations
151
+
152
+ - `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
153
+ been observed 0-based in Databricks (serverless, October 2026). dataowl keeps the raw
154
+ value in `ColumnInfo.position` and shows a running number in the `#` column.
155
+
156
+ ## License
157
+
158
+ MIT. Copyright (c) 2026 Alexandro Dronnen. See [LICENSE](LICENSE).
@@ -0,0 +1,128 @@
1
+ # dataowl
2
+
3
+ dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
4
+ rows, schema and table properties. It is meant for data engineers who build a dbt staging
5
+ model and want the facts without writing exploratory SQL. It reports what exists and never
6
+ gives recommendations.
7
+
8
+ ## Requirements
9
+
10
+ - Unity Catalog.
11
+ - Databricks Runtime 14.3 LTS or above, or serverless compute.
12
+ Verified on serverless (Spark Connect).
13
+ - PySpark is provided by Databricks. dataowl has no other runtime dependencies.
14
+
15
+ ## Installation
16
+
17
+ In a Databricks notebook:
18
+
19
+ ```python
20
+ %pip install dataowl==0.1.0
21
+ ```
22
+
23
+ ## Usage
24
+
25
+ ```python
26
+ import dataowl
27
+
28
+ dataowl.inspect("catalog.schema.table").show()
29
+ ```
30
+
31
+ As a JSON-serializable dict, where every fact has a value, a source and a reason when it
32
+ is unavailable:
33
+
34
+ ```python
35
+ overview = dataowl.inspect("catalog.schema.table")
36
+ overview.to_dict()
37
+ ```
38
+
39
+ Row counts are skipped for views, materialized views and foreign tables unless you ask for
40
+ them:
41
+
42
+ ```python
43
+ dataowl.inspect("catalog.schema.some_view", count_views=True).show()
44
+ ```
45
+
46
+ Names with characters other than letters, digits and underscore are quoted with backticks:
47
+ `` dataowl.inspect("`my-catalog`.schema.table") ``.
48
+
49
+ ## Example output
50
+
51
+ Output for Databricks' public sample data (`samples.nyctaxi.trips`):
52
+
53
+ ```text
54
+ samples.nyctaxi.trips (MANAGED, DELTA)
55
+
56
+ Size: 354.2 KB
57
+ Files: 1 (avg 354.2 KB)
58
+ Rows: 21 932
59
+ Columns: 6 (6 incl. nested)
60
+ Partitioned by: –
61
+ Clustered by: –
62
+ Change Data Feed: true
63
+ Log retention: not set (default)
64
+ Deleted file retention: not set (default)
65
+ Created: 2025-09-30 11:28
66
+ Last modified: 2026-09-14 15:07
67
+ Owner: System user
68
+ Comment: –
69
+
70
+ SCHEMA
71
+ # name type nullable comment
72
+ 1 tpep_pickup_datetime timestamp yes
73
+ 2 tpep_dropoff_datetime timestamp yes
74
+ 3 trip_distance double yes
75
+ 4 fare_amount double yes
76
+ 5 pickup_zip int yes
77
+ 6 dropoff_zip int yes
78
+ ```
79
+
80
+ ## Principles
81
+
82
+ - **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
83
+ analyzed.
84
+ - **Facts only.** No recommendations or assessments.
85
+ - **No row values leave Spark.** Only aggregates and metadata are collected.
86
+ - **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
87
+ object type make the affected facts `n/a (<reason>)`; the rest of the report is still
88
+ produced. Only a table that does not exist or is not accessible raises
89
+ `TableNotFoundError`.
90
+
91
+ ## Facts per object type
92
+
93
+ | Facts | Source | MANAGED, EXTERNAL | STREAMING_TABLE, MATERIALIZED_VIEW | FOREIGN | VIEW | UNKNOWN or n/a type |
94
+ |---|---|---|---|---|---|---|
95
+ | Object type, format, owner, comment, created | `information_schema.tables` | yes | yes | yes | yes | yes (n/a if the query fails) |
96
+ | Size, files, avg file size, last modified, partitioning, clustering | `DESCRIBE DETAIL` | yes | attempted¹ | attempted¹ | no | attempted¹ |
97
+ | Columns and schema | `information_schema.columns` | yes | yes | yes | yes | yes |
98
+ | Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
99
+ | Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
100
+ | Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
101
+
102
+ ¹ The query runs. If Databricks does not support it for the object, the facts are shown as
103
+ `n/a (<reason>)`.
104
+
105
+ "UNKNOWN" is a `table_type` dataowl does not recognize, such as `MANAGED_SHALLOW_CLONE`. The
106
+ original value is shown in the header. "n/a type" means the object type could not be read.
107
+
108
+ ## Fields
109
+
110
+ - **Last modified** is the last data change, from `DESCRIBE DETAIL`. **Created** is from
111
+ `information_schema.tables`.
112
+ - **Rows** is an exact `COUNT(*)`. On large tables this costs compute.
113
+ - **Files (avg ...)** is the size divided by the number of files.
114
+ - **Columns (... incl. nested)** counts every field, including fields inside structs, array
115
+ elements and map keys and values.
116
+ - **not set (default)** means the table property is not set, so Databricks uses its default.
117
+ dataowl does not report what the default is.
118
+ - **#** in the schema table is a running number from 1 in column order.
119
+
120
+ ## Known deviations
121
+
122
+ - `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
123
+ been observed 0-based in Databricks (serverless, October 2026). dataowl keeps the raw
124
+ value in `ColumnInfo.position` and shows a running number in the `#` column.
125
+
126
+ ## License
127
+
128
+ MIT. Copyright (c) 2026 Alexandro Dronnen. See [LICENSE](LICENSE).
@@ -0,0 +1,70 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "dataowl"
7
+ version = "0.1.0"
8
+ description = "Facts about Databricks data products for building dbt staging models."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Alexandro Dronnen" }]
14
+ keywords = ["databricks", "unity-catalog", "delta-lake", "dbt", "data-engineering"]
15
+ classifiers = [
16
+ "Development Status :: 3 - Alpha",
17
+ "Intended Audience :: Developers",
18
+ "Operating System :: OS Independent",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3 :: Only",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Topic :: Database",
25
+ "Typing :: Typed",
26
+ ]
27
+ dependencies = []
28
+
29
+ [project.urls]
30
+ Homepage = "https://github.com/alexD1990/dataowl"
31
+ Repository = "https://github.com/alexD1990/dataowl"
32
+ Changelog = "https://github.com/alexD1990/dataowl/blob/master/CHANGELOG.md"
33
+ Issues = "https://github.com/alexD1990/dataowl/issues"
34
+
35
+ [project.optional-dependencies]
36
+ dev = ["pytest", "ruff==0.16.9", "pyspark>=3.4", "mypy==2.3.1"]
37
+
38
+ [tool.hatch.build.targets.wheel]
39
+ packages = ["src/dataowl"]
40
+
41
+ [tool.hatch.build.targets.sdist]
42
+ include = ["src", "tests", "README.md", "LICENSE", "CHANGELOG.md", "pyproject.toml"]
43
+ exclude = ["build_plan.md", ".github"]
44
+
45
+ [tool.ruff]
46
+ line-length = 100
47
+ target-version = "py310"
48
+ extend-exclude = ["build_plan.md"]
49
+
50
+ [tool.ruff.lint]
51
+ select = ["E", "F", "I", "UP", "B", "SIM"]
52
+
53
+ [tool.ruff.lint.isort]
54
+ required-imports = ["from __future__ import annotations"]
55
+
56
+ [tool.mypy]
57
+ files = ["src"]
58
+ python_version = "3.10"
59
+ disallow_untyped_defs = true
60
+ disallow_incomplete_defs = true
61
+ check_untyped_defs = true
62
+ warn_return_any = true
63
+ warn_unused_ignores = true
64
+ warn_redundant_casts = true
65
+ warn_unreachable = true
66
+ strict_equality = true
67
+
68
+ [tool.pytest.ini_options]
69
+ testpaths = ["tests/unit"]
70
+ norecursedirs = ["tests/integration"]
@@ -0,0 +1,62 @@
1
+ """dataowl: facts about Databricks data products for building dbt staging models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING
6
+
7
+ from dataowl.collect.columns import collect_columns
8
+ from dataowl.collect.counts import collect_row_count
9
+ from dataowl.collect.detail import collect_detail
10
+ from dataowl.collect.properties import collect_properties
11
+ from dataowl.collect.tables import collect_table_info
12
+ from dataowl.errors import TableNotFoundError
13
+ from dataowl.identifiers import parse_table
14
+ from dataowl.model.overview import Overview
15
+ from dataowl.runner import get_runner
16
+
17
+ if TYPE_CHECKING:
18
+ from pyspark.sql import SparkSession
19
+
20
+ __all__ = ["Overview", "TableNotFoundError", "inspect"]
21
+
22
+
23
+ def inspect(
24
+ table: str, *, spark: SparkSession | None = None, count_views: bool = False
25
+ ) -> Overview:
26
+ """Collect overview facts for a table.
27
+
28
+ Raises ValueError for an invalid table name, RuntimeError when no SparkSession is
29
+ available, and TableNotFoundError when the table does not exist or is not accessible.
30
+ Every other failure makes the affected facts unavailable.
31
+ """
32
+ ref = parse_table(table)
33
+ runner = get_runner(spark)
34
+
35
+ info = collect_table_info(runner, ref)
36
+ detail = collect_detail(runner, ref, info.object_type)
37
+ columns = collect_columns(runner, ref)
38
+ properties = collect_properties(runner, ref, info.object_type)
39
+ num_rows = collect_row_count(runner, ref, info.object_type, count_views=count_views)
40
+
41
+ return Overview(
42
+ table=ref,
43
+ object_type=info.object_type,
44
+ object_type_raw=info.table_type_raw,
45
+ format=info.format,
46
+ owner=info.owner,
47
+ comment=info.comment,
48
+ created=info.created,
49
+ last_modified=detail.last_modified,
50
+ size_bytes=detail.size_bytes,
51
+ num_files=detail.num_files,
52
+ avg_file_size_bytes=detail.avg_file_size_bytes,
53
+ num_rows=num_rows,
54
+ num_columns=columns.num_columns,
55
+ num_fields_nested=columns.num_fields_nested,
56
+ partition_columns=detail.partition_columns,
57
+ clustering_columns=detail.clustering_columns,
58
+ change_data_feed=properties.change_data_feed,
59
+ log_retention=properties.log_retention,
60
+ deleted_file_retention=properties.deleted_file_retention,
61
+ columns=columns.columns,
62
+ )
@@ -0,0 +1,40 @@
1
+ """Collection layer: queries to raw data to model objects. The only layer that knows SQL."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from dataowl.model.facts import Fact
8
+ from dataowl.model.overview import ObjectType
9
+ from dataowl.runner import SqlRunner
10
+
11
+ _MAX_REASON_LENGTH = 200
12
+
13
+ NOT_FOR_VIEWS = "Not available for views"
14
+
15
+
16
+ def is_view(object_type: Fact[ObjectType]) -> bool:
17
+ """True only when the object type is known to be VIEW."""
18
+ return object_type.available and object_type.value is ObjectType.VIEW
19
+
20
+
21
+ def safe_query(
22
+ runner: SqlRunner, sql: str, params: dict[str, Any] | None = None
23
+ ) -> list[dict[str, Any]] | Exception:
24
+ """Run a query and return the rows, or the exception instead of raising it."""
25
+ try:
26
+ return runner.query(sql, params)
27
+ except Exception as exc:
28
+ return exc
29
+
30
+
31
+ def short_reason(exc: Exception) -> str:
32
+ """First line of the error message, at most 200 characters.
33
+
34
+ Falls back to the exception type name when the message is empty.
35
+ """
36
+ lines = str(exc).strip().splitlines()
37
+ reason = lines[0].strip() if lines else type(exc).__name__
38
+ if len(reason) > _MAX_REASON_LENGTH:
39
+ reason = reason[: _MAX_REASON_LENGTH - 1] + "…"
40
+ return reason
@@ -0,0 +1,135 @@
1
+ """Schema facts from information_schema.columns and the Spark schema."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING, Any, TypeVar
6
+
7
+ from dataowl.collect import safe_query, short_reason
8
+ from dataowl.identifiers import TableRef
9
+ from dataowl.model.facts import Fact
10
+ from dataowl.model.overview import ColumnInfo, ColumnsInfo
11
+ from dataowl.runner import SqlRunner
12
+
13
+ if TYPE_CHECKING:
14
+ from pyspark.sql.types import DataType, StructType
15
+
16
+ T = TypeVar("T")
17
+
18
+ NO_COLUMNS = "information_schema.columns returned no rows"
19
+
20
+ _QUERY = """
21
+ SELECT
22
+ column_name AS column_name,
23
+ ordinal_position AS ordinal_position,
24
+ full_data_type AS full_data_type,
25
+ data_type AS data_type,
26
+ is_nullable AS is_nullable,
27
+ comment AS comment
28
+ FROM {catalog}.information_schema.columns
29
+ WHERE table_schema = :schema
30
+ AND table_name = :table
31
+ ORDER BY ordinal_position
32
+ """
33
+
34
+
35
+ def collect_columns(runner: SqlRunner, ref: TableRef) -> ColumnsInfo:
36
+ """Read top-level columns from information_schema and count nested fields from the schema.
37
+
38
+ The two sources fail independently.
39
+ """
40
+ columns, num_columns = _top_level_columns(runner, ref)
41
+ return ColumnsInfo(
42
+ columns=columns,
43
+ num_columns=num_columns,
44
+ num_fields_nested=_nested_field_count(runner, ref),
45
+ )
46
+
47
+
48
+ def count_fields(schema: StructType) -> int:
49
+ """Count all fields in a schema, including nested fields.
50
+
51
+ Every StructField counts 1. Fields inside a struct, inside the element type of an
52
+ array and inside the key and value types of a map are added recursively.
53
+
54
+ Example: ``a INT, b STRUCT<x INT, y ARRAY<STRUCT<z INT>>>, m MAP<STRING, STRUCT<v INT>>``
55
+ counts a, b, x, y, z, m and v, which gives 7.
56
+ """
57
+ from pyspark.sql.types import ArrayType, MapType, StructType
58
+
59
+ def nested(data_type: DataType) -> int:
60
+ if isinstance(data_type, StructType):
61
+ return sum(1 + nested(field.dataType) for field in data_type.fields)
62
+ if isinstance(data_type, ArrayType):
63
+ return nested(data_type.elementType)
64
+ if isinstance(data_type, MapType):
65
+ return nested(data_type.keyType) + nested(data_type.valueType)
66
+ return 0
67
+
68
+ return nested(schema)
69
+
70
+
71
+ def _top_level_columns(
72
+ runner: SqlRunner, ref: TableRef
73
+ ) -> tuple[Fact[tuple[ColumnInfo, ...]], Fact[int]]:
74
+ result = safe_query(
75
+ runner,
76
+ _QUERY.format(catalog=ref.quoted_catalog()),
77
+ # Unity Catalog stores names in lower case. Lowering the parameters instead of
78
+ # the columns keeps the WHERE clause free of functions on information_schema columns.
79
+ {"schema": ref.schema.lower(), "table": ref.table.lower()},
80
+ )
81
+ if isinstance(result, Exception):
82
+ reason = short_reason(result)
83
+ return Fact.unavailable("metadata", reason), Fact.unavailable("metadata", reason)
84
+ if not result:
85
+ return Fact.unavailable("metadata", NO_COLUMNS), Fact.unavailable("metadata", NO_COLUMNS)
86
+
87
+ try:
88
+ columns = tuple(_column(row) for row in result)
89
+ except ValueError as exc:
90
+ reason = f"Unexpected row format in information_schema.columns: {exc}"
91
+ return Fact.unavailable("metadata", reason), Fact.unavailable("metadata", reason)
92
+ return Fact(columns, source="metadata"), Fact(len(columns), source="metadata")
93
+
94
+
95
+ def _column(row: dict[str, Any]) -> ColumnInfo:
96
+ """Convert one row. Raises ValueError if a key is missing or has an unexpected type."""
97
+ name = _get(row, "column_name", str)
98
+ position = _get(row, "ordinal_position", int)
99
+ full_data_type = row.get("full_data_type")
100
+ if full_data_type is None:
101
+ data_type = _get(row, "data_type", str)
102
+ elif isinstance(full_data_type, str):
103
+ data_type = full_data_type
104
+ else:
105
+ raise ValueError(f"full_data_type has type {type(full_data_type).__name__}")
106
+ is_nullable = _get(row, "is_nullable", str)
107
+ if is_nullable not in ("YES", "NO"):
108
+ raise ValueError(f"is_nullable has value {is_nullable!r}")
109
+ comment = row.get("comment")
110
+ if comment is not None and not isinstance(comment, str):
111
+ raise ValueError(f"comment has type {type(comment).__name__}")
112
+ return ColumnInfo(
113
+ name=name,
114
+ position=position,
115
+ data_type=data_type,
116
+ nullable=is_nullable == "YES",
117
+ comment=comment,
118
+ )
119
+
120
+
121
+ def _get(row: dict[str, Any], key: str, expected: type[T]) -> T:
122
+ if key not in row:
123
+ raise ValueError(f"missing {key}")
124
+ value = row[key]
125
+ if not isinstance(value, expected) or isinstance(value, bool):
126
+ raise ValueError(f"{key} has type {type(value).__name__}")
127
+ return value
128
+
129
+
130
+ def _nested_field_count(runner: SqlRunner, ref: TableRef) -> Fact[int]:
131
+ try:
132
+ schema = runner.schema(ref)
133
+ except Exception as exc:
134
+ return Fact.unavailable("metadata", short_reason(exc))
135
+ return Fact(count_fields(schema), source="metadata")