dataowl 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. dataowl-0.2.0/CHANGELOG.md +62 -0
  2. dataowl-0.2.0/PKG-INFO +345 -0
  3. dataowl-0.2.0/README.md +315 -0
  4. {dataowl-0.1.0 → dataowl-0.2.0}/pyproject.toml +1 -1
  5. dataowl-0.2.0/src/dataowl/__init__.py +127 -0
  6. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/collect/columns.py +22 -9
  7. dataowl-0.2.0/src/dataowl/collect/compare.py +110 -0
  8. dataowl-0.2.0/src/dataowl/collect/constraints.py +78 -0
  9. dataowl-0.2.0/src/dataowl/collect/history.py +124 -0
  10. dataowl-0.2.0/src/dataowl/collect/keys.py +205 -0
  11. dataowl-0.2.0/src/dataowl/collect/timestamps.py +395 -0
  12. dataowl-0.2.0/src/dataowl/model/analysis.py +132 -0
  13. dataowl-0.2.0/src/dataowl/model/history.py +19 -0
  14. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/model/overview.py +18 -0
  15. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/render/format.py +20 -1
  16. dataowl-0.2.0/src/dataowl/render/terminal.py +387 -0
  17. dataowl-0.2.0/tests/fixtures/analysis_days.txt +17 -0
  18. dataowl-0.2.0/tests/fixtures/analysis_full.txt +60 -0
  19. dataowl-0.2.0/tests/fixtures/analysis_minimal.txt +7 -0
  20. dataowl-0.2.0/tests/fixtures/analysis_unavailable.txt +35 -0
  21. dataowl-0.2.0/tests/fixtures/describe_history_table.json +263 -0
  22. {dataowl-0.1.0 → dataowl-0.2.0}/tests/fixtures/overview_table.txt +5 -0
  23. {dataowl-0.1.0 → dataowl-0.2.0}/tests/fixtures/overview_unavailable.txt +3 -0
  24. {dataowl-0.1.0 → dataowl-0.2.0}/tests/fixtures/overview_view.txt +3 -0
  25. {dataowl-0.1.0 → dataowl-0.2.0}/tests/integration/setup_test_tables.sql +205 -11
  26. dataowl-0.2.0/tests/integration/test_integration.py +475 -0
  27. dataowl-0.2.0/tests/unit/test_analyze.py +375 -0
  28. dataowl-0.2.0/tests/unit/test_collect_compare.py +211 -0
  29. dataowl-0.2.0/tests/unit/test_collect_constraints.py +113 -0
  30. dataowl-0.2.0/tests/unit/test_collect_history.py +338 -0
  31. dataowl-0.2.0/tests/unit/test_collect_keys.py +416 -0
  32. dataowl-0.2.0/tests/unit/test_collect_timestamps.py +704 -0
  33. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_format.py +48 -2
  34. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_inspect.py +64 -2
  35. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_terminal.py +87 -1
  36. dataowl-0.2.0/tests/unit/test_terminal_analysis.py +349 -0
  37. dataowl-0.1.0/CHANGELOG.md +0 -32
  38. dataowl-0.1.0/PKG-INFO +0 -158
  39. dataowl-0.1.0/README.md +0 -128
  40. dataowl-0.1.0/src/dataowl/__init__.py +0 -62
  41. dataowl-0.1.0/src/dataowl/collect/history.py +0 -1
  42. dataowl-0.1.0/src/dataowl/collect/keys.py +0 -1
  43. dataowl-0.1.0/src/dataowl/collect/timestamps.py +0 -1
  44. dataowl-0.1.0/src/dataowl/model/analysis.py +0 -1
  45. dataowl-0.1.0/src/dataowl/model/history.py +0 -1
  46. dataowl-0.1.0/src/dataowl/render/terminal.py +0 -138
  47. dataowl-0.1.0/tests/integration/test_integration.py +0 -208
  48. {dataowl-0.1.0 → dataowl-0.2.0}/.gitignore +0 -0
  49. {dataowl-0.1.0 → dataowl-0.2.0}/LICENSE +0 -0
  50. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/collect/__init__.py +0 -0
  51. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/collect/counts.py +0 -0
  52. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/collect/detail.py +0 -0
  53. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/collect/properties.py +0 -0
  54. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/collect/tables.py +0 -0
  55. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/errors.py +0 -0
  56. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/identifiers.py +0 -0
  57. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/model/__init__.py +0 -0
  58. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/model/facts.py +0 -0
  59. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/py.typed +0 -0
  60. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/render/__init__.py +0 -0
  61. {dataowl-0.1.0 → dataowl-0.2.0}/src/dataowl/runner.py +0 -0
  62. {dataowl-0.1.0 → dataowl-0.2.0}/tests/conftest.py +0 -0
  63. {dataowl-0.1.0 → dataowl-0.2.0}/tests/fixtures/describe_detail_table.json +0 -0
  64. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_collect.py +0 -0
  65. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_collect_columns.py +0 -0
  66. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_collect_counts.py +0 -0
  67. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_collect_detail.py +0 -0
  68. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_collect_properties.py +0 -0
  69. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_collect_tables.py +0 -0
  70. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_facts.py +0 -0
  71. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_fake_runner.py +0 -0
  72. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_identifiers.py +0 -0
  73. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_import.py +0 -0
  74. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_import_boundaries.py +0 -0
  75. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_no_judgement.py +0 -0
  76. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_read_only.py +0 -0
  77. {dataowl-0.1.0 → dataowl-0.2.0}/tests/unit/test_runner.py +0 -0
@@ -0,0 +1,62 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ ## [Unreleased]
6
+
7
+ ## [0.2.0] - 2026-10-08
8
+
9
+ Column analysis (`analyze`) and an excerpt of the Delta history in `inspect`.
10
+
11
+ ### Added
12
+ - `dataowl.analyze(table, *, key=None, timestamp=None, compare=None, days=30, spark=None)`
13
+ returning a `ColumnAnalysis` with `to_dict()` and `show(show_days=False)`. Only the
14
+ columns given by the user are analyzed, and at least one of `key`, `timestamp` and
15
+ `compare` is required.
16
+ - Key analysis (`key`): rows, rows with null in the key, distinct keys, keys occurring more
17
+ than once and their rows, median and maximum rows per key, and the number of keys with 1,
18
+ 2–10, 11–100 and more than 100 rows. Single columns, composite keys and lists of both.
19
+ - Declared primary key from `information_schema.table_constraints` and
20
+ `information_schema.key_column_usage`, collected when `key` is given. Shown as
21
+ informational and not enforced.
22
+ - Time column analysis (`timestamp`) for `TIMESTAMP`, `TIMESTAMP_NTZ` and `DATE`: min, max,
23
+ nulls and their share, values after the current time, distinct dates, gaps in whole days
24
+ between distinct dates, and rows per day for the last `days` whole days without today.
25
+ - Comparison of two time columns (`compare`): rows where the second column is after, equal
26
+ to or before the first, and rows where either is null.
27
+ - `show(show_days=True)` prints one line per day in the rows-per-day window.
28
+ - History excerpt in `inspect` from `DESCRIBE HISTORY`: new `Overview` fields
29
+ `history_first_commit`, `history_last_commit`, `history_num_commits` and
30
+ `history_operations`. `WRITE` with mode `Overwrite` is counted as `WRITE (overwrite)`.
31
+ Not run for views.
32
+
33
+ ### Changed
34
+ - The `inspect` output has a HISTORY block between the summary and the schema, with the
35
+ observation window of the history.
36
+
37
+ ## [0.1.0] - 2026-10-08
38
+
39
+ First release: overview facts for a table (`inspect`).
40
+
41
+ ### Added
42
+ - `dataowl.inspect(table, *, spark=None, count_views=False)` returning an `Overview` with
43
+ `to_dict()` and `show()`.
44
+ - Facts from `information_schema.tables`: object type (with the raw `table_type`), format,
45
+ owner, comment and created time.
46
+ - Facts from `DESCRIBE DETAIL`: size, number of files, average file size, last data
47
+ modification, partition and clustering columns. Not run for views.
48
+ - Schema from `information_schema.columns`, and the field count including nested fields
49
+ from the Spark schema.
50
+ - Exact row count with `COUNT(*)`. Skipped for views, materialized views, foreign tables and
51
+ objects whose type cannot be read, unless `count_views=True`.
52
+ - Change Data Feed, log retention and deleted file retention from `SHOW TBLPROPERTIES`. Not
53
+ run for views.
54
+ - Every fact carries its source (`metadata`, `exact`, `derived`). Facts that cannot be
55
+ collected are marked unavailable with a reason; only a missing table raises
56
+ (`TableNotFoundError`).
57
+ - Terminal rendering of the overview.
58
+
59
+ ### Known deviations
60
+ - `information_schema.columns.ordinal_position` is observed 0-based in Databricks, although
61
+ the documentation says it is numbered from 1. `ColumnInfo.position` keeps the raw value;
62
+ the `#` column in the output is a running number from 1.
dataowl-0.2.0/PKG-INFO ADDED
@@ -0,0 +1,345 @@
1
+ Metadata-Version: 2.5
2
+ Name: dataowl
3
+ Version: 0.2.0
4
+ Summary: Facts about Databricks data products for building dbt staging models.
5
+ Project-URL: Homepage, https://github.com/alexD1990/dataowl
6
+ Project-URL: Repository, https://github.com/alexD1990/dataowl
7
+ Project-URL: Changelog, https://github.com/alexD1990/dataowl/blob/master/CHANGELOG.md
8
+ Project-URL: Issues, https://github.com/alexD1990/dataowl/issues
9
+ Author: Alexandro Dronnen
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: data-engineering,databricks,dbt,delta-lake,unity-catalog
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3 :: Only
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Database
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.10
24
+ Provides-Extra: dev
25
+ Requires-Dist: mypy==2.3.1; extra == 'dev'
26
+ Requires-Dist: pyspark>=3.4; extra == 'dev'
27
+ Requires-Dist: pytest; extra == 'dev'
28
+ Requires-Dist: ruff==0.16.9; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # dataowl
32
+
33
+ dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
34
+ rows, schema, table properties and an excerpt of the Delta history, and facts about the
35
+ columns you choose: grain and uniqueness of keys, time span and cadence of time columns, and
36
+ how two time columns compare. It is meant for data engineers who build a dbt staging model
37
+ and want the facts without writing exploratory SQL. It reports what exists and never gives
38
+ recommendations.
39
+
40
+ ## Requirements
41
+
42
+ - Unity Catalog.
43
+ - Databricks Runtime 14.3 LTS or above, or serverless compute.
44
+ Verified on serverless (Spark Connect).
45
+ - PySpark is provided by Databricks. dataowl has no other runtime dependencies.
46
+
47
+ ## Installation
48
+
49
+ In a Databricks notebook:
50
+
51
+ ```python
52
+ %pip install dataowl==0.2.0
53
+ ```
54
+
55
+ ## Usage
56
+
57
+ dataowl works in two steps: `inspect` gives an overview of the table, and `analyze` gives
58
+ facts about columns you choose.
59
+
60
+ ### inspect
61
+
62
+ ```python
63
+ import dataowl
64
+
65
+ dataowl.inspect("catalog.schema.table").show()
66
+ ```
67
+
68
+ As a JSON-serializable dict, where every fact has a value, a source and a reason when it
69
+ is unavailable:
70
+
71
+ ```python
72
+ overview = dataowl.inspect("catalog.schema.table")
73
+ overview.to_dict()
74
+ ```
75
+
76
+ Row counts are skipped for views, materialized views and foreign tables unless you ask for
77
+ them:
78
+
79
+ ```python
80
+ dataowl.inspect("catalog.schema.some_view", count_views=True).show()
81
+ ```
82
+
83
+ Names with characters other than letters, digits and underscore are quoted with backticks:
84
+ `` dataowl.inspect("`my-catalog`.schema.table") ``.
85
+
86
+ The overview includes an excerpt of the Delta history from `DESCRIBE HISTORY`: the oldest and
87
+ newest commit, the number of commits and the number of commits per operation. The header of
88
+ the HISTORY block always shows the observation window (oldest – newest commit), because an
89
+ operation that is missing is only missing for the period the history covers:
90
+
91
+ - The history is limited by `delta.logRetentionDuration` (see **Log retention**). Older
92
+ commits are not included.
93
+ - `WRITE (overwrite)` is a `WRITE` commit with mode `Overwrite`. It also includes partial
94
+ overwrites, for example with `replaceWhere`.
95
+ - `CREATE OR REPLACE TABLE AS SELECT`, which dbt uses for the `table` materialization, is
96
+ shown under its own operation name, not as `WRITE (overwrite)`.
97
+
98
+ ### analyze
99
+
100
+ ```python
101
+ analysis = dataowl.analyze(
102
+ "catalog.schema.table",
103
+ key=["customer_id", ("customer_id", "order_id")],
104
+ timestamp="updated_at",
105
+ compare=("created_at", "updated_at"),
106
+ )
107
+ analysis.show()
108
+ ```
109
+
110
+ Full signature:
111
+
112
+ ```python
113
+ analyze(
114
+ table: str,
115
+ *,
116
+ key: str | tuple[str, ...] | list[str | tuple[str, ...]] | None = None,
117
+ timestamp: str | list[str] | None = None,
118
+ compare: tuple[str, str] | None = None,
119
+ days: int = 30,
120
+ spark: SparkSession | None = None,
121
+ ) -> ColumnAnalysis
122
+ ```
123
+
124
+ At least one of `key`, `timestamp` and `compare` must be given. Only the columns you give are
125
+ analyzed; dataowl never chooses or suggests columns. Each key, each timestamp column and the
126
+ comparison run their own queries against the table's data, so every column or combination
127
+ you add costs compute. Column names are matched case insensitively.
128
+
129
+ - **key** analyzes grain and uniqueness. A string is one column, a tuple is one composite
130
+ key, and a list holds several keys of either kind, each analyzed separately:
131
+ `key=["customer_id", ("customer_id", "order_id")]` analyzes `customer_id` on its own and
132
+ the combination of `customer_id` and `order_id`. MAP columns cannot be used.
133
+ - **Primary key.** When `key` is given, dataowl also reads the declared primary key from
134
+ `information_schema`. A primary key in Databricks is informational and not enforced. In
135
+ `ColumnAnalysis.primary_key`, `None` means it was not collected (no `key`), a fact with
136
+ value `None` means no primary key is declared, and an unavailable fact has a reason.
137
+ - **timestamp** analyzes one column or a list of columns of type `TIMESTAMP`,
138
+ `TIMESTAMP_NTZ` or `DATE`, each separately: min, max, nulls, values after the current
139
+ time, distinct dates, the gaps in whole days between distinct dates, and rows per day.
140
+ Other types, including dates stored as `STRING`, raise `ValueError`.
141
+ - **days** (default 30) sets the window for rows per day: the last `days` whole days, not
142
+ including today, in the session time zone. Days without rows count as 0.
143
+ - **compare** counts the rows where the second column is after, equal to or before the
144
+ first, and the rows where at least one of them is null. Columns of different types are
145
+ compared with Spark's implicit type conversion: a `DATE` is compared as that date at
146
+ 00:00:00, so a `DATE` on the same day as a `TIMESTAMP` later than midnight counts as
147
+ before it.
148
+
149
+ ```python
150
+ analysis.show(show_days=True) # also prints one line per day in the window
151
+ analysis.to_dict() # JSON-serializable, like Overview.to_dict()
152
+ ```
153
+
154
+ `analyze` raises:
155
+
156
+ - `ValueError` for an invalid table name, when `key`, `timestamp` and `compare` are all
157
+ `None`, for invalid `days`, and for an unknown or unsupported column. All input is
158
+ validated before any query against the table's data.
159
+ - `RuntimeError` when no SparkSession is available, and when the schema cannot be read from
160
+ `information_schema.columns`, since the columns cannot then be validated.
161
+ - `TableNotFoundError` when the table does not exist or is not accessible.
162
+
163
+ Every other failure makes the affected facts `n/a (<reason>)`.
164
+
165
+ ## Example output
166
+
167
+ Output for Databricks' public sample data (`samples.nyctaxi.trips`).
168
+
169
+ ```python
170
+ dataowl.inspect("samples.nyctaxi.trips").show()
171
+ ```
172
+
173
+ ```text
174
+ samples.nyctaxi.trips (MANAGED, DELTA)
175
+
176
+ Size: 354.2 KB
177
+ Files: 1 (avg 354.2 KB)
178
+ Rows: 21 932
179
+ Columns: 6 (6 incl. nested)
180
+ Partitioned by: –
181
+ Clustered by: –
182
+ Change Data Feed: true
183
+ Log retention: not set (default)
184
+ Deleted file retention: not set (default)
185
+ Created: 2025-09-30 11:28
186
+ Last modified: 2026-09-14 15:07
187
+ Owner: System user
188
+ Comment: –
189
+
190
+ HISTORY (2025-09-09 15:05 – 2026-09-14 15:07, 223 commits)
191
+ CREATE OR REPLACE TABLE AS SELECT: 212
192
+ SET TBLPROPERTIES: 11
193
+
194
+ SCHEMA
195
+ # name type nullable comment
196
+ 1 tpep_pickup_datetime timestamp yes
197
+ 2 tpep_dropoff_datetime timestamp yes
198
+ 3 trip_distance double yes
199
+ 4 fare_amount double yes
200
+ 5 pickup_zip int yes
201
+ 6 dropoff_zip int yes
202
+ ```
203
+
204
+ ```python
205
+ dataowl.analyze(
206
+ "samples.nyctaxi.trips",
207
+ key=["pickup_zip", ("pickup_zip", "dropoff_zip")],
208
+ timestamp="tpep_pickup_datetime",
209
+ compare=("tpep_pickup_datetime", "tpep_dropoff_datetime"),
210
+ ).show()
211
+ ```
212
+
213
+ ```text
214
+ samples.nyctaxi.trips
215
+
216
+ PRIMARY KEY none declared
217
+
218
+ KEY pickup_zip
219
+ Rows: 21 932
220
+ Rows with null in key: 0
221
+ Distinct keys: 128
222
+ Keys occurring >1 time: 100
223
+ Rows in those keys: 21 904
224
+ Rows per key: median 19 · max 1 227
225
+ 1 row: 28
226
+ 2–10 rows: 27
227
+ 11–100 rows: 31
228
+ >100 rows: 42
229
+
230
+ KEY pickup_zip + dropoff_zip
231
+ Rows: 21 932
232
+ Rows with null in key: 0
233
+ Distinct keys: 3 369
234
+ Keys occurring >1 time: 2 060
235
+ Rows in those keys: 20 623
236
+ Rows per key: median 2 · max 143
237
+ 1 row: 1 309
238
+ 2–10 rows: 1 533
239
+ 11–100 rows: 518
240
+ >100 rows: 9
241
+
242
+ TIMESTAMP tpep_pickup_datetime
243
+ Min: 2016-01-01 00:04
244
+ Max: 2016-02-29 23:51
245
+ Nulls: 0 (0.00 %)
246
+ Future values: 0
247
+ Distinct dates: 60
248
+ Days between distinct dates:
249
+ min 1 · median 1 · mean 1 · max 1
250
+ Rows per day, last 30 complete days (2026-09-08 – 2026-10-07):
251
+ median 0 · mean 0 · min 0 · max 0
252
+ Note: gaps are measured in whole days; cadence below one day is not visible.
253
+ Note: counts reflect current values of the column, not historical changes.
254
+
255
+ COMPARE tpep_pickup_datetime → tpep_dropoff_datetime
256
+ tpep_dropoff_datetime > tpep_pickup_datetime: 21 931
257
+ tpep_dropoff_datetime = tpep_pickup_datetime: 1
258
+ tpep_dropoff_datetime < tpep_pickup_datetime: 0
259
+ Either is null: 0
260
+ ```
261
+
262
+ The data in `samples.nyctaxi.trips` is from January and February 2016, so every day in the
263
+ rows-per-day window has 0 rows.
264
+
265
+ ## Principles
266
+
267
+ - **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
268
+ analyzed.
269
+ - **Facts only.** No recommendations or assessments.
270
+ - **No row values leave Spark.** Only aggregates and metadata are collected. Min and max of
271
+ time columns are the only exception.
272
+ - **You choose the columns.** `analyze` only analyzes the columns you give.
273
+ - **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
274
+ object type make the affected facts `n/a (<reason>)`; the rest of the report is still
275
+ produced. Only a table that does not exist or is not accessible raises
276
+ `TableNotFoundError`.
277
+
278
+ ## Facts per object type
279
+
280
+ | Facts | Source | MANAGED, EXTERNAL | STREAMING_TABLE, MATERIALIZED_VIEW | FOREIGN | VIEW | UNKNOWN or n/a type |
281
+ |---|---|---|---|---|---|---|
282
+ | Object type, format, owner, comment, created | `information_schema.tables` | yes | yes | yes | yes | yes (n/a if the query fails) |
283
+ | Size, files, avg file size, last modified, partitioning, clustering | `DESCRIBE DETAIL` | yes | attempted¹ | attempted¹ | no | attempted¹ |
284
+ | Columns and schema | `information_schema.columns` | yes | yes | yes | yes | yes |
285
+ | Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
286
+ | Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
287
+ | History excerpt | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
288
+ | Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
289
+
290
+ ¹ The query runs. If Databricks does not support it for the object, the facts are shown as
291
+ `n/a (<reason>)`.
292
+
293
+ "UNKNOWN" is a `table_type` dataowl does not recognize, such as `MANAGED_SHALLOW_CLONE`. The
294
+ original value is shown in the header. "n/a type" means the object type could not be read.
295
+
296
+ ## Fields
297
+
298
+ ### inspect
299
+
300
+ - **Last modified** is the last data change, from `DESCRIBE DETAIL`. **Created** is from
301
+ `information_schema.tables`.
302
+ - **Rows** is an exact `COUNT(*)`. On large tables this costs compute.
303
+ - **Files (avg ...)** is the size divided by the number of files.
304
+ - **Columns (... incl. nested)** counts every field, including fields inside structs, array
305
+ elements and map keys and values.
306
+ - **not set (default)** means the table property is not set, so Databricks uses its default.
307
+ dataowl does not report what the default is.
308
+ - **#** in the schema table is a running number from 1 in column order.
309
+ - **HISTORY** shows the oldest and newest commit in the available history, the number of
310
+ commits, and the number of commits per operation.
311
+
312
+ ### analyze
313
+
314
+ - **PRIMARY KEY** is the declared primary key, in declared order. It is informational and
315
+ not enforced by Databricks. The line is left out when `key` is not given.
316
+ - **KEY ... Rows** is every row in the table, including rows with null in a key column.
317
+ - **Rows with null in key** counts rows where at least one key column is null. These rows
318
+ are left out of every other line in the KEY block.
319
+ - **Distinct keys** is the number of distinct key values without null.
320
+ - **Keys occurring >1 time** is the number of key values that occur in more than one row.
321
+ **Rows in those keys** is the number of rows that have one of those key values.
322
+ - **Rows per key** is the median and the maximum number of rows per key value, followed by
323
+ the number of key values with 1, 2–10, 11–100 and more than 100 rows.
324
+ - **TIMESTAMP ... Nulls** is the number of rows with null, and their share of all rows.
325
+ - **Future values** counts values after the current time of the session. A `DATE` is
326
+ compared as that date at 00:00:00.
327
+ - **Days between distinct dates** is measured in whole days between consecutive distinct
328
+ dates. Cadence below one day is not visible.
329
+ - **Rows per day** covers the last `days` whole days without today, in the session time
330
+ zone, and counts current values of the column, not historical changes.
331
+ - **COMPARE first → second** counts rows where the second column is after (`>`), equal to
332
+ (`=`) or before (`<`) the first. Rows with null in either column are counted only in
333
+ **Either is null**, so the four numbers add up to all rows.
334
+ - Medians and means are rounded to one decimal and shown without decimals when the rounded
335
+ value is whole.
336
+
337
+ ## Known deviations
338
+
339
+ - `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
340
+ been observed 0-based in Databricks (serverless, October 2026). dataowl keeps the raw
341
+ value in `ColumnInfo.position` and shows a running number in the `#` column.
342
+
343
+ ## License
344
+
345
+ MIT. Copyright (c) 2026 Alexandro Dronnen. See [LICENSE](LICENSE).