dataowl 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dataowl-0.1.0/.gitignore +10 -0
- dataowl-0.1.0/CHANGELOG.md +32 -0
- dataowl-0.1.0/LICENSE +21 -0
- dataowl-0.1.0/PKG-INFO +158 -0
- dataowl-0.1.0/README.md +128 -0
- dataowl-0.1.0/pyproject.toml +70 -0
- dataowl-0.1.0/src/dataowl/__init__.py +62 -0
- dataowl-0.1.0/src/dataowl/collect/__init__.py +40 -0
- dataowl-0.1.0/src/dataowl/collect/columns.py +135 -0
- dataowl-0.1.0/src/dataowl/collect/counts.py +46 -0
- dataowl-0.1.0/src/dataowl/collect/detail.py +80 -0
- dataowl-0.1.0/src/dataowl/collect/history.py +1 -0
- dataowl-0.1.0/src/dataowl/collect/keys.py +1 -0
- dataowl-0.1.0/src/dataowl/collect/properties.py +75 -0
- dataowl-0.1.0/src/dataowl/collect/tables.py +80 -0
- dataowl-0.1.0/src/dataowl/collect/timestamps.py +1 -0
- dataowl-0.1.0/src/dataowl/errors.py +7 -0
- dataowl-0.1.0/src/dataowl/identifiers.py +112 -0
- dataowl-0.1.0/src/dataowl/model/__init__.py +3 -0
- dataowl-0.1.0/src/dataowl/model/analysis.py +1 -0
- dataowl-0.1.0/src/dataowl/model/facts.py +75 -0
- dataowl-0.1.0/src/dataowl/model/history.py +1 -0
- dataowl-0.1.0/src/dataowl/model/overview.py +130 -0
- dataowl-0.1.0/src/dataowl/py.typed +0 -0
- dataowl-0.1.0/src/dataowl/render/__init__.py +3 -0
- dataowl-0.1.0/src/dataowl/render/format.py +44 -0
- dataowl-0.1.0/src/dataowl/render/terminal.py +138 -0
- dataowl-0.1.0/src/dataowl/runner.py +47 -0
- dataowl-0.1.0/tests/conftest.py +113 -0
- dataowl-0.1.0/tests/fixtures/describe_detail_table.json +20 -0
- dataowl-0.1.0/tests/fixtures/overview_table.txt +22 -0
- dataowl-0.1.0/tests/fixtures/overview_unavailable.txt +18 -0
- dataowl-0.1.0/tests/fixtures/overview_view.txt +20 -0
- dataowl-0.1.0/tests/integration/setup_test_tables.sql +289 -0
- dataowl-0.1.0/tests/integration/test_integration.py +208 -0
- dataowl-0.1.0/tests/unit/test_collect.py +44 -0
- dataowl-0.1.0/tests/unit/test_collect_columns.py +299 -0
- dataowl-0.1.0/tests/unit/test_collect_counts.py +102 -0
- dataowl-0.1.0/tests/unit/test_collect_detail.py +148 -0
- dataowl-0.1.0/tests/unit/test_collect_properties.py +178 -0
- dataowl-0.1.0/tests/unit/test_collect_tables.py +144 -0
- dataowl-0.1.0/tests/unit/test_facts.py +135 -0
- dataowl-0.1.0/tests/unit/test_fake_runner.py +81 -0
- dataowl-0.1.0/tests/unit/test_format.py +62 -0
- dataowl-0.1.0/tests/unit/test_identifiers.py +167 -0
- dataowl-0.1.0/tests/unit/test_import.py +7 -0
- dataowl-0.1.0/tests/unit/test_import_boundaries.py +31 -0
- dataowl-0.1.0/tests/unit/test_inspect.py +298 -0
- dataowl-0.1.0/tests/unit/test_no_judgement.py +35 -0
- dataowl-0.1.0/tests/unit/test_read_only.py +65 -0
- dataowl-0.1.0/tests/unit/test_runner.py +106 -0
- dataowl-0.1.0/tests/unit/test_terminal.py +201 -0
dataowl-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented in this file.
|
|
4
|
+
|
|
5
|
+
## [Unreleased]
|
|
6
|
+
|
|
7
|
+
## [0.1.0] - 2026-10-08
|
|
8
|
+
|
|
9
|
+
First release: overview facts for a table (`inspect`).
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- `dataowl.inspect(table, *, spark=None, count_views=False)` returning an `Overview` with
|
|
13
|
+
`to_dict()` and `show()`.
|
|
14
|
+
- Facts from `information_schema.tables`: object type (with the raw `table_type`), format,
|
|
15
|
+
owner, comment and created time.
|
|
16
|
+
- Facts from `DESCRIBE DETAIL`: size, number of files, average file size, last data
|
|
17
|
+
modification, partition and clustering columns. Not run for views.
|
|
18
|
+
- Schema from `information_schema.columns`, and the field count including nested fields
|
|
19
|
+
from the Spark schema.
|
|
20
|
+
- Exact row count with `COUNT(*)`. Skipped for views, materialized views, foreign tables and
|
|
21
|
+
objects whose type cannot be read, unless `count_views=True`.
|
|
22
|
+
- Change Data Feed, log retention and deleted file retention from `SHOW TBLPROPERTIES`. Not
|
|
23
|
+
run for views.
|
|
24
|
+
- Every fact carries its source (`metadata`, `exact`, `derived`). Facts that cannot be
|
|
25
|
+
collected are marked unavailable with a reason; only a missing table raises
|
|
26
|
+
(`TableNotFoundError`).
|
|
27
|
+
- Terminal rendering of the overview.
|
|
28
|
+
|
|
29
|
+
### Known deviations
|
|
30
|
+
- `information_schema.columns.ordinal_position` is observed 0-based in Databricks, although
|
|
31
|
+
the documentation says it is numbered from 1. `ColumnInfo.position` keeps the raw value;
|
|
32
|
+
the `#` column in the output is a running number from 1.
|
dataowl-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Alexandro Dronnen
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
dataowl-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: dataowl
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Facts about Databricks data products for building dbt staging models.
|
|
5
|
+
Project-URL: Homepage, https://github.com/alexD1990/dataowl
|
|
6
|
+
Project-URL: Repository, https://github.com/alexD1990/dataowl
|
|
7
|
+
Project-URL: Changelog, https://github.com/alexD1990/dataowl/blob/master/CHANGELOG.md
|
|
8
|
+
Project-URL: Issues, https://github.com/alexD1990/dataowl/issues
|
|
9
|
+
Author: Alexandro Dronnen
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: data-engineering,databricks,dbt,delta-lake,unity-catalog
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Database
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: mypy==2.3.1; extra == 'dev'
|
|
26
|
+
Requires-Dist: pyspark>=3.4; extra == 'dev'
|
|
27
|
+
Requires-Dist: pytest; extra == 'dev'
|
|
28
|
+
Requires-Dist: ruff==0.16.9; extra == 'dev'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# dataowl
|
|
32
|
+
|
|
33
|
+
dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
|
|
34
|
+
rows, schema and table properties. It is meant for data engineers who build a dbt staging
|
|
35
|
+
model and want the facts without writing exploratory SQL. It reports what exists and never
|
|
36
|
+
gives recommendations.
|
|
37
|
+
|
|
38
|
+
## Requirements
|
|
39
|
+
|
|
40
|
+
- Unity Catalog.
|
|
41
|
+
- Databricks Runtime 14.3 LTS or above, or serverless compute.
|
|
42
|
+
Verified on serverless (Spark Connect).
|
|
43
|
+
- PySpark is provided by Databricks. dataowl has no other runtime dependencies.
|
|
44
|
+
|
|
45
|
+
## Installation
|
|
46
|
+
|
|
47
|
+
In a Databricks notebook:
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
%pip install dataowl==0.1.0
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Usage
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
import dataowl
|
|
57
|
+
|
|
58
|
+
dataowl.inspect("catalog.schema.table").show()
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
As a JSON-serializable dict, where every fact has a value, a source and a reason when it
|
|
62
|
+
is unavailable:
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
overview = dataowl.inspect("catalog.schema.table")
|
|
66
|
+
overview.to_dict()
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Row counts are skipped for views, materialized views and foreign tables unless you ask for
|
|
70
|
+
them:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
dataowl.inspect("catalog.schema.some_view", count_views=True).show()
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Names with characters other than letters, digits and underscore are quoted with backticks:
|
|
77
|
+
`` dataowl.inspect("`my-catalog`.schema.table") ``.
|
|
78
|
+
|
|
79
|
+
## Example output
|
|
80
|
+
|
|
81
|
+
Output for Databricks' public sample data (`samples.nyctaxi.trips`):
|
|
82
|
+
|
|
83
|
+
```text
|
|
84
|
+
samples.nyctaxi.trips (MANAGED, DELTA)
|
|
85
|
+
|
|
86
|
+
Size: 354.2 KB
|
|
87
|
+
Files: 1 (avg 354.2 KB)
|
|
88
|
+
Rows: 21 932
|
|
89
|
+
Columns: 6 (6 incl. nested)
|
|
90
|
+
Partitioned by: –
|
|
91
|
+
Clustered by: –
|
|
92
|
+
Change Data Feed: true
|
|
93
|
+
Log retention: not set (default)
|
|
94
|
+
Deleted file retention: not set (default)
|
|
95
|
+
Created: 2025-09-30 11:28
|
|
96
|
+
Last modified: 2026-09-14 15:07
|
|
97
|
+
Owner: System user
|
|
98
|
+
Comment: –
|
|
99
|
+
|
|
100
|
+
SCHEMA
|
|
101
|
+
# name type nullable comment
|
|
102
|
+
1 tpep_pickup_datetime timestamp yes
|
|
103
|
+
2 tpep_dropoff_datetime timestamp yes
|
|
104
|
+
3 trip_distance double yes
|
|
105
|
+
4 fare_amount double yes
|
|
106
|
+
5 pickup_zip int yes
|
|
107
|
+
6 dropoff_zip int yes
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## Principles
|
|
111
|
+
|
|
112
|
+
- **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
|
|
113
|
+
analyzed.
|
|
114
|
+
- **Facts only.** No recommendations or assessments.
|
|
115
|
+
- **No row values leave Spark.** Only aggregates and metadata are collected.
|
|
116
|
+
- **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
|
|
117
|
+
object type make the affected facts `n/a (<reason>)`; the rest of the report is still
|
|
118
|
+
produced. Only a table that does not exist or is not accessible raises
|
|
119
|
+
`TableNotFoundError`.
|
|
120
|
+
|
|
121
|
+
## Facts per object type
|
|
122
|
+
|
|
123
|
+
| Facts | Source | MANAGED, EXTERNAL | STREAMING_TABLE, MATERIALIZED_VIEW | FOREIGN | VIEW | UNKNOWN or n/a type |
|
|
124
|
+
|---|---|---|---|---|---|---|
|
|
125
|
+
| Object type, format, owner, comment, created | `information_schema.tables` | yes | yes | yes | yes | yes (n/a if the query fails) |
|
|
126
|
+
| Size, files, avg file size, last modified, partitioning, clustering | `DESCRIBE DETAIL` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
127
|
+
| Columns and schema | `information_schema.columns` | yes | yes | yes | yes | yes |
|
|
128
|
+
| Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
|
|
129
|
+
| Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
130
|
+
| Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
|
|
131
|
+
|
|
132
|
+
¹ The query runs. If Databricks does not support it for the object, the facts are shown as
|
|
133
|
+
`n/a (<reason>)`.
|
|
134
|
+
|
|
135
|
+
"UNKNOWN" is a `table_type` dataowl does not recognize, such as `MANAGED_SHALLOW_CLONE`. The
|
|
136
|
+
original value is shown in the header. "n/a type" means the object type could not be read.
|
|
137
|
+
|
|
138
|
+
## Fields
|
|
139
|
+
|
|
140
|
+
- **Last modified** is the last data change, from `DESCRIBE DETAIL`. **Created** is from
|
|
141
|
+
`information_schema.tables`.
|
|
142
|
+
- **Rows** is an exact `COUNT(*)`. On large tables this costs compute.
|
|
143
|
+
- **Files (avg ...)** is the size divided by the number of files.
|
|
144
|
+
- **Columns (... incl. nested)** counts every field, including fields inside structs, array
|
|
145
|
+
elements and map keys and values.
|
|
146
|
+
- **not set (default)** means the table property is not set, so Databricks uses its default.
|
|
147
|
+
dataowl does not report what the default is.
|
|
148
|
+
- **#** in the schema table is a running number from 1 in column order.
|
|
149
|
+
|
|
150
|
+
## Known deviations
|
|
151
|
+
|
|
152
|
+
- `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
|
|
153
|
+
been observed 0-based in Databricks (serverless, October 2026). dataowl keeps the raw
|
|
154
|
+
value in `ColumnInfo.position` and shows a running number in the `#` column.
|
|
155
|
+
|
|
156
|
+
## License
|
|
157
|
+
|
|
158
|
+
MIT. Copyright (c) 2026 Alexandro Dronnen. See [LICENSE](LICENSE).
|
dataowl-0.1.0/README.md
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
# dataowl
|
|
2
|
+
|
|
3
|
+
dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
|
|
4
|
+
rows, schema and table properties. It is meant for data engineers who build a dbt staging
|
|
5
|
+
model and want the facts without writing exploratory SQL. It reports what exists and never
|
|
6
|
+
gives recommendations.
|
|
7
|
+
|
|
8
|
+
## Requirements
|
|
9
|
+
|
|
10
|
+
- Unity Catalog.
|
|
11
|
+
- Databricks Runtime 14.3 LTS or above, or serverless compute.
|
|
12
|
+
Verified on serverless (Spark Connect).
|
|
13
|
+
- PySpark is provided by Databricks. dataowl has no other runtime dependencies.
|
|
14
|
+
|
|
15
|
+
## Installation
|
|
16
|
+
|
|
17
|
+
In a Databricks notebook:
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
%pip install dataowl==0.1.0
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Usage
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
import dataowl
|
|
27
|
+
|
|
28
|
+
dataowl.inspect("catalog.schema.table").show()
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
As a JSON-serializable dict, where every fact has a value, a source and a reason when it
|
|
32
|
+
is unavailable:
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
overview = dataowl.inspect("catalog.schema.table")
|
|
36
|
+
overview.to_dict()
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Row counts are skipped for views, materialized views and foreign tables unless you ask for
|
|
40
|
+
them:
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
dataowl.inspect("catalog.schema.some_view", count_views=True).show()
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Names with characters other than letters, digits and underscore are quoted with backticks:
|
|
47
|
+
`` dataowl.inspect("`my-catalog`.schema.table") ``.
|
|
48
|
+
|
|
49
|
+
## Example output
|
|
50
|
+
|
|
51
|
+
Output for Databricks' public sample data (`samples.nyctaxi.trips`):
|
|
52
|
+
|
|
53
|
+
```text
|
|
54
|
+
samples.nyctaxi.trips (MANAGED, DELTA)
|
|
55
|
+
|
|
56
|
+
Size: 354.2 KB
|
|
57
|
+
Files: 1 (avg 354.2 KB)
|
|
58
|
+
Rows: 21 932
|
|
59
|
+
Columns: 6 (6 incl. nested)
|
|
60
|
+
Partitioned by: –
|
|
61
|
+
Clustered by: –
|
|
62
|
+
Change Data Feed: true
|
|
63
|
+
Log retention: not set (default)
|
|
64
|
+
Deleted file retention: not set (default)
|
|
65
|
+
Created: 2025-09-30 11:28
|
|
66
|
+
Last modified: 2026-09-14 15:07
|
|
67
|
+
Owner: System user
|
|
68
|
+
Comment: –
|
|
69
|
+
|
|
70
|
+
SCHEMA
|
|
71
|
+
# name type nullable comment
|
|
72
|
+
1 tpep_pickup_datetime timestamp yes
|
|
73
|
+
2 tpep_dropoff_datetime timestamp yes
|
|
74
|
+
3 trip_distance double yes
|
|
75
|
+
4 fare_amount double yes
|
|
76
|
+
5 pickup_zip int yes
|
|
77
|
+
6 dropoff_zip int yes
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Principles
|
|
81
|
+
|
|
82
|
+
- **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
|
|
83
|
+
analyzed.
|
|
84
|
+
- **Facts only.** No recommendations or assessments.
|
|
85
|
+
- **No row values leave Spark.** Only aggregates and metadata are collected.
|
|
86
|
+
- **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
|
|
87
|
+
object type make the affected facts `n/a (<reason>)`; the rest of the report is still
|
|
88
|
+
produced. Only a table that does not exist or is not accessible raises
|
|
89
|
+
`TableNotFoundError`.
|
|
90
|
+
|
|
91
|
+
## Facts per object type
|
|
92
|
+
|
|
93
|
+
| Facts | Source | MANAGED, EXTERNAL | STREAMING_TABLE, MATERIALIZED_VIEW | FOREIGN | VIEW | UNKNOWN or n/a type |
|
|
94
|
+
|---|---|---|---|---|---|---|
|
|
95
|
+
| Object type, format, owner, comment, created | `information_schema.tables` | yes | yes | yes | yes | yes (n/a if the query fails) |
|
|
96
|
+
| Size, files, avg file size, last modified, partitioning, clustering | `DESCRIBE DETAIL` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
97
|
+
| Columns and schema | `information_schema.columns` | yes | yes | yes | yes | yes |
|
|
98
|
+
| Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
|
|
99
|
+
| Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
100
|
+
| Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
|
|
101
|
+
|
|
102
|
+
¹ The query runs. If Databricks does not support it for the object, the facts are shown as
|
|
103
|
+
`n/a (<reason>)`.
|
|
104
|
+
|
|
105
|
+
"UNKNOWN" is a `table_type` dataowl does not recognize, such as `MANAGED_SHALLOW_CLONE`. The
|
|
106
|
+
original value is shown in the header. "n/a type" means the object type could not be read.
|
|
107
|
+
|
|
108
|
+
## Fields
|
|
109
|
+
|
|
110
|
+
- **Last modified** is the last data change, from `DESCRIBE DETAIL`. **Created** is from
|
|
111
|
+
`information_schema.tables`.
|
|
112
|
+
- **Rows** is an exact `COUNT(*)`. On large tables this costs compute.
|
|
113
|
+
- **Files (avg ...)** is the size divided by the number of files.
|
|
114
|
+
- **Columns (... incl. nested)** counts every field, including fields inside structs, array
|
|
115
|
+
elements and map keys and values.
|
|
116
|
+
- **not set (default)** means the table property is not set, so Databricks uses its default.
|
|
117
|
+
dataowl does not report what the default is.
|
|
118
|
+
- **#** in the schema table is a running number from 1 in column order.
|
|
119
|
+
|
|
120
|
+
## Known deviations
|
|
121
|
+
|
|
122
|
+
- `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
|
|
123
|
+
been observed 0-based in Databricks (serverless, October 2026). dataowl keeps the raw
|
|
124
|
+
value in `ColumnInfo.position` and shows a running number in the `#` column.
|
|
125
|
+
|
|
126
|
+
## License
|
|
127
|
+
|
|
128
|
+
MIT. Copyright (c) 2026 Alexandro Dronnen. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dataowl"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Facts about Databricks data products for building dbt staging models."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Alexandro Dronnen" }]
|
|
14
|
+
keywords = ["databricks", "unity-catalog", "delta-lake", "dbt", "data-engineering"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Topic :: Database",
|
|
25
|
+
"Typing :: Typed",
|
|
26
|
+
]
|
|
27
|
+
dependencies = []
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/alexD1990/dataowl"
|
|
31
|
+
Repository = "https://github.com/alexD1990/dataowl"
|
|
32
|
+
Changelog = "https://github.com/alexD1990/dataowl/blob/master/CHANGELOG.md"
|
|
33
|
+
Issues = "https://github.com/alexD1990/dataowl/issues"
|
|
34
|
+
|
|
35
|
+
[project.optional-dependencies]
|
|
36
|
+
dev = ["pytest", "ruff==0.16.9", "pyspark>=3.4", "mypy==2.3.1"]
|
|
37
|
+
|
|
38
|
+
[tool.hatch.build.targets.wheel]
|
|
39
|
+
packages = ["src/dataowl"]
|
|
40
|
+
|
|
41
|
+
[tool.hatch.build.targets.sdist]
|
|
42
|
+
include = ["src", "tests", "README.md", "LICENSE", "CHANGELOG.md", "pyproject.toml"]
|
|
43
|
+
exclude = ["build_plan.md", ".github"]
|
|
44
|
+
|
|
45
|
+
[tool.ruff]
|
|
46
|
+
line-length = 100
|
|
47
|
+
target-version = "py310"
|
|
48
|
+
extend-exclude = ["build_plan.md"]
|
|
49
|
+
|
|
50
|
+
[tool.ruff.lint]
|
|
51
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
52
|
+
|
|
53
|
+
[tool.ruff.lint.isort]
|
|
54
|
+
required-imports = ["from __future__ import annotations"]
|
|
55
|
+
|
|
56
|
+
[tool.mypy]
|
|
57
|
+
files = ["src"]
|
|
58
|
+
python_version = "3.10"
|
|
59
|
+
disallow_untyped_defs = true
|
|
60
|
+
disallow_incomplete_defs = true
|
|
61
|
+
check_untyped_defs = true
|
|
62
|
+
warn_return_any = true
|
|
63
|
+
warn_unused_ignores = true
|
|
64
|
+
warn_redundant_casts = true
|
|
65
|
+
warn_unreachable = true
|
|
66
|
+
strict_equality = true
|
|
67
|
+
|
|
68
|
+
[tool.pytest.ini_options]
|
|
69
|
+
testpaths = ["tests/unit"]
|
|
70
|
+
norecursedirs = ["tests/integration"]
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""dataowl: facts about Databricks data products for building dbt staging models."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING
|
|
6
|
+
|
|
7
|
+
from dataowl.collect.columns import collect_columns
|
|
8
|
+
from dataowl.collect.counts import collect_row_count
|
|
9
|
+
from dataowl.collect.detail import collect_detail
|
|
10
|
+
from dataowl.collect.properties import collect_properties
|
|
11
|
+
from dataowl.collect.tables import collect_table_info
|
|
12
|
+
from dataowl.errors import TableNotFoundError
|
|
13
|
+
from dataowl.identifiers import parse_table
|
|
14
|
+
from dataowl.model.overview import Overview
|
|
15
|
+
from dataowl.runner import get_runner
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
from pyspark.sql import SparkSession
|
|
19
|
+
|
|
20
|
+
__all__ = ["Overview", "TableNotFoundError", "inspect"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def inspect(
|
|
24
|
+
table: str, *, spark: SparkSession | None = None, count_views: bool = False
|
|
25
|
+
) -> Overview:
|
|
26
|
+
"""Collect overview facts for a table.
|
|
27
|
+
|
|
28
|
+
Raises ValueError for an invalid table name, RuntimeError when no SparkSession is
|
|
29
|
+
available, and TableNotFoundError when the table does not exist or is not accessible.
|
|
30
|
+
Every other failure makes the affected facts unavailable.
|
|
31
|
+
"""
|
|
32
|
+
ref = parse_table(table)
|
|
33
|
+
runner = get_runner(spark)
|
|
34
|
+
|
|
35
|
+
info = collect_table_info(runner, ref)
|
|
36
|
+
detail = collect_detail(runner, ref, info.object_type)
|
|
37
|
+
columns = collect_columns(runner, ref)
|
|
38
|
+
properties = collect_properties(runner, ref, info.object_type)
|
|
39
|
+
num_rows = collect_row_count(runner, ref, info.object_type, count_views=count_views)
|
|
40
|
+
|
|
41
|
+
return Overview(
|
|
42
|
+
table=ref,
|
|
43
|
+
object_type=info.object_type,
|
|
44
|
+
object_type_raw=info.table_type_raw,
|
|
45
|
+
format=info.format,
|
|
46
|
+
owner=info.owner,
|
|
47
|
+
comment=info.comment,
|
|
48
|
+
created=info.created,
|
|
49
|
+
last_modified=detail.last_modified,
|
|
50
|
+
size_bytes=detail.size_bytes,
|
|
51
|
+
num_files=detail.num_files,
|
|
52
|
+
avg_file_size_bytes=detail.avg_file_size_bytes,
|
|
53
|
+
num_rows=num_rows,
|
|
54
|
+
num_columns=columns.num_columns,
|
|
55
|
+
num_fields_nested=columns.num_fields_nested,
|
|
56
|
+
partition_columns=detail.partition_columns,
|
|
57
|
+
clustering_columns=detail.clustering_columns,
|
|
58
|
+
change_data_feed=properties.change_data_feed,
|
|
59
|
+
log_retention=properties.log_retention,
|
|
60
|
+
deleted_file_retention=properties.deleted_file_retention,
|
|
61
|
+
columns=columns.columns,
|
|
62
|
+
)
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Collection layer: queries to raw data to model objects. The only layer that knows SQL."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from dataowl.model.facts import Fact
|
|
8
|
+
from dataowl.model.overview import ObjectType
|
|
9
|
+
from dataowl.runner import SqlRunner
|
|
10
|
+
|
|
11
|
+
_MAX_REASON_LENGTH = 200
|
|
12
|
+
|
|
13
|
+
NOT_FOR_VIEWS = "Not available for views"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def is_view(object_type: Fact[ObjectType]) -> bool:
|
|
17
|
+
"""True only when the object type is known to be VIEW."""
|
|
18
|
+
return object_type.available and object_type.value is ObjectType.VIEW
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def safe_query(
|
|
22
|
+
runner: SqlRunner, sql: str, params: dict[str, Any] | None = None
|
|
23
|
+
) -> list[dict[str, Any]] | Exception:
|
|
24
|
+
"""Run a query and return the rows, or the exception instead of raising it."""
|
|
25
|
+
try:
|
|
26
|
+
return runner.query(sql, params)
|
|
27
|
+
except Exception as exc:
|
|
28
|
+
return exc
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def short_reason(exc: Exception) -> str:
|
|
32
|
+
"""First line of the error message, at most 200 characters.
|
|
33
|
+
|
|
34
|
+
Falls back to the exception type name when the message is empty.
|
|
35
|
+
"""
|
|
36
|
+
lines = str(exc).strip().splitlines()
|
|
37
|
+
reason = lines[0].strip() if lines else type(exc).__name__
|
|
38
|
+
if len(reason) > _MAX_REASON_LENGTH:
|
|
39
|
+
reason = reason[: _MAX_REASON_LENGTH - 1] + "…"
|
|
40
|
+
return reason
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Schema facts from information_schema.columns and the Spark schema."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING, Any, TypeVar
|
|
6
|
+
|
|
7
|
+
from dataowl.collect import safe_query, short_reason
|
|
8
|
+
from dataowl.identifiers import TableRef
|
|
9
|
+
from dataowl.model.facts import Fact
|
|
10
|
+
from dataowl.model.overview import ColumnInfo, ColumnsInfo
|
|
11
|
+
from dataowl.runner import SqlRunner
|
|
12
|
+
|
|
13
|
+
if TYPE_CHECKING:
|
|
14
|
+
from pyspark.sql.types import DataType, StructType
|
|
15
|
+
|
|
16
|
+
T = TypeVar("T")
|
|
17
|
+
|
|
18
|
+
NO_COLUMNS = "information_schema.columns returned no rows"
|
|
19
|
+
|
|
20
|
+
_QUERY = """
|
|
21
|
+
SELECT
|
|
22
|
+
column_name AS column_name,
|
|
23
|
+
ordinal_position AS ordinal_position,
|
|
24
|
+
full_data_type AS full_data_type,
|
|
25
|
+
data_type AS data_type,
|
|
26
|
+
is_nullable AS is_nullable,
|
|
27
|
+
comment AS comment
|
|
28
|
+
FROM {catalog}.information_schema.columns
|
|
29
|
+
WHERE table_schema = :schema
|
|
30
|
+
AND table_name = :table
|
|
31
|
+
ORDER BY ordinal_position
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def collect_columns(runner: SqlRunner, ref: TableRef) -> ColumnsInfo:
|
|
36
|
+
"""Read top-level columns from information_schema and count nested fields from the schema.
|
|
37
|
+
|
|
38
|
+
The two sources fail independently.
|
|
39
|
+
"""
|
|
40
|
+
columns, num_columns = _top_level_columns(runner, ref)
|
|
41
|
+
return ColumnsInfo(
|
|
42
|
+
columns=columns,
|
|
43
|
+
num_columns=num_columns,
|
|
44
|
+
num_fields_nested=_nested_field_count(runner, ref),
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def count_fields(schema: StructType) -> int:
|
|
49
|
+
"""Count all fields in a schema, including nested fields.
|
|
50
|
+
|
|
51
|
+
Every StructField counts 1. Fields inside a struct, inside the element type of an
|
|
52
|
+
array and inside the key and value types of a map are added recursively.
|
|
53
|
+
|
|
54
|
+
Example: ``a INT, b STRUCT<x INT, y ARRAY<STRUCT<z INT>>>, m MAP<STRING, STRUCT<v INT>>``
|
|
55
|
+
counts a, b, x, y, z, m and v, which gives 7.
|
|
56
|
+
"""
|
|
57
|
+
from pyspark.sql.types import ArrayType, MapType, StructType
|
|
58
|
+
|
|
59
|
+
def nested(data_type: DataType) -> int:
|
|
60
|
+
if isinstance(data_type, StructType):
|
|
61
|
+
return sum(1 + nested(field.dataType) for field in data_type.fields)
|
|
62
|
+
if isinstance(data_type, ArrayType):
|
|
63
|
+
return nested(data_type.elementType)
|
|
64
|
+
if isinstance(data_type, MapType):
|
|
65
|
+
return nested(data_type.keyType) + nested(data_type.valueType)
|
|
66
|
+
return 0
|
|
67
|
+
|
|
68
|
+
return nested(schema)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _top_level_columns(
|
|
72
|
+
runner: SqlRunner, ref: TableRef
|
|
73
|
+
) -> tuple[Fact[tuple[ColumnInfo, ...]], Fact[int]]:
|
|
74
|
+
result = safe_query(
|
|
75
|
+
runner,
|
|
76
|
+
_QUERY.format(catalog=ref.quoted_catalog()),
|
|
77
|
+
# Unity Catalog stores names in lower case. Lowering the parameters instead of
|
|
78
|
+
# the columns keeps the WHERE clause free of functions on information_schema columns.
|
|
79
|
+
{"schema": ref.schema.lower(), "table": ref.table.lower()},
|
|
80
|
+
)
|
|
81
|
+
if isinstance(result, Exception):
|
|
82
|
+
reason = short_reason(result)
|
|
83
|
+
return Fact.unavailable("metadata", reason), Fact.unavailable("metadata", reason)
|
|
84
|
+
if not result:
|
|
85
|
+
return Fact.unavailable("metadata", NO_COLUMNS), Fact.unavailable("metadata", NO_COLUMNS)
|
|
86
|
+
|
|
87
|
+
try:
|
|
88
|
+
columns = tuple(_column(row) for row in result)
|
|
89
|
+
except ValueError as exc:
|
|
90
|
+
reason = f"Unexpected row format in information_schema.columns: {exc}"
|
|
91
|
+
return Fact.unavailable("metadata", reason), Fact.unavailable("metadata", reason)
|
|
92
|
+
return Fact(columns, source="metadata"), Fact(len(columns), source="metadata")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _column(row: dict[str, Any]) -> ColumnInfo:
|
|
96
|
+
"""Convert one row. Raises ValueError if a key is missing or has an unexpected type."""
|
|
97
|
+
name = _get(row, "column_name", str)
|
|
98
|
+
position = _get(row, "ordinal_position", int)
|
|
99
|
+
full_data_type = row.get("full_data_type")
|
|
100
|
+
if full_data_type is None:
|
|
101
|
+
data_type = _get(row, "data_type", str)
|
|
102
|
+
elif isinstance(full_data_type, str):
|
|
103
|
+
data_type = full_data_type
|
|
104
|
+
else:
|
|
105
|
+
raise ValueError(f"full_data_type has type {type(full_data_type).__name__}")
|
|
106
|
+
is_nullable = _get(row, "is_nullable", str)
|
|
107
|
+
if is_nullable not in ("YES", "NO"):
|
|
108
|
+
raise ValueError(f"is_nullable has value {is_nullable!r}")
|
|
109
|
+
comment = row.get("comment")
|
|
110
|
+
if comment is not None and not isinstance(comment, str):
|
|
111
|
+
raise ValueError(f"comment has type {type(comment).__name__}")
|
|
112
|
+
return ColumnInfo(
|
|
113
|
+
name=name,
|
|
114
|
+
position=position,
|
|
115
|
+
data_type=data_type,
|
|
116
|
+
nullable=is_nullable == "YES",
|
|
117
|
+
comment=comment,
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _get(row: dict[str, Any], key: str, expected: type[T]) -> T:
|
|
122
|
+
if key not in row:
|
|
123
|
+
raise ValueError(f"missing {key}")
|
|
124
|
+
value = row[key]
|
|
125
|
+
if not isinstance(value, expected) or isinstance(value, bool):
|
|
126
|
+
raise ValueError(f"{key} has type {type(value).__name__}")
|
|
127
|
+
return value
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _nested_field_count(runner: SqlRunner, ref: TableRef) -> Fact[int]:
|
|
131
|
+
try:
|
|
132
|
+
schema = runner.schema(ref)
|
|
133
|
+
except Exception as exc:
|
|
134
|
+
return Fact.unavailable("metadata", short_reason(exc))
|
|
135
|
+
return Fact(count_fields(schema), source="metadata")
|