dataowl 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataowl-0.2.0 → dataowl-0.3.0}/CHANGELOG.md +26 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/PKG-INFO +140 -8
- {dataowl-0.2.0 → dataowl-0.3.0}/README.md +139 -7
- {dataowl-0.2.0 → dataowl-0.3.0}/pyproject.toml +1 -1
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/__init__.py +46 -6
- dataowl-0.3.0/src/dataowl/collect/__init__.py +80 -0
- dataowl-0.3.0/src/dataowl/collect/history.py +424 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/tables.py +8 -3
- dataowl-0.3.0/src/dataowl/model/history.py +126 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/render/terminal.py +108 -0
- dataowl-0.3.0/tests/fixtures/describe_history_metrics.json +83 -0
- dataowl-0.3.0/tests/fixtures/history_full.txt +31 -0
- dataowl-0.3.0/tests/fixtures/history_limit.txt +23 -0
- dataowl-0.3.0/tests/fixtures/history_partial.txt +23 -0
- dataowl-0.3.0/tests/fixtures/history_view.txt +8 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/integration/setup_test_tables.sql +20 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/integration/test_integration.py +134 -2
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_analyze.py +14 -0
- dataowl-0.3.0/tests/unit/test_collect.py +147 -0
- dataowl-0.3.0/tests/unit/test_collect_history.py +999 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_tables.py +35 -0
- dataowl-0.3.0/tests/unit/test_history.py +232 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_inspect.py +16 -0
- dataowl-0.3.0/tests/unit/test_terminal_history.py +147 -0
- dataowl-0.2.0/src/dataowl/collect/__init__.py +0 -40
- dataowl-0.2.0/src/dataowl/collect/history.py +0 -124
- dataowl-0.2.0/src/dataowl/model/history.py +0 -19
- dataowl-0.2.0/tests/unit/test_collect.py +0 -44
- dataowl-0.2.0/tests/unit/test_collect_history.py +0 -338
- {dataowl-0.2.0 → dataowl-0.3.0}/.gitignore +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/LICENSE +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/columns.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/compare.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/constraints.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/counts.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/detail.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/keys.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/properties.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/timestamps.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/errors.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/identifiers.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/__init__.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/analysis.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/facts.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/overview.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/py.typed +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/render/__init__.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/render/format.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/runner.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/conftest.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_days.txt +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_full.txt +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_minimal.txt +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_unavailable.txt +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/describe_detail_table.json +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/describe_history_table.json +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/overview_table.txt +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/overview_unavailable.txt +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/overview_view.txt +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_columns.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_compare.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_constraints.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_counts.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_detail.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_keys.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_properties.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_timestamps.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_facts.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_fake_runner.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_format.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_identifiers.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_import.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_import_boundaries.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_no_judgement.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_read_only.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_runner.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_terminal.py +0 -0
- {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_terminal_analysis.py +0 -0
|
@@ -4,6 +4,32 @@ All notable changes to this project are documented in this file.
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.3.0] - 2026-10-08
|
|
8
|
+
|
|
9
|
+
Write behaviour from the Delta history (`history`).
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- `dataowl.history(table, *, spark=None, limit=None)` returning a `HistoryAnalysis` with
|
|
13
|
+
`to_dict()` and `show()`. `limit` reads only the newest `limit` commits.
|
|
14
|
+
- Observation window (oldest and newest commit, calendar days, commits), commits per day
|
|
15
|
+
(median, min, max, with 0 for days without commits), commits per hour of day, and
|
|
16
|
+
commits per operation with `WRITE (overwrite)` as its own category.
|
|
17
|
+
- Inserted, updated and deleted rows per operation from `operationMetrics`
|
|
18
|
+
(`numTargetRowsInserted`, `numTargetRowsUpdated` and `numTargetRowsDeleted` for `MERGE`,
|
|
19
|
+
`numOutputRows` for `WRITE`, `WRITE (overwrite)` and `... AS SELECT`, `numUpdatedRows`
|
|
20
|
+
for `UPDATE`, `numDeletedRows` for `DELETE`): commits with the metric out of all, sum,
|
|
21
|
+
median and max.
|
|
22
|
+
- The session time zone (`current_timezone()`) as a fact. Days and hours are counted in
|
|
23
|
+
that time zone.
|
|
24
|
+
- Notes in the output: the history is limited by `delta.logRetentionDuration`, rows
|
|
25
|
+
replaced by `WRITE (overwrite)` are not counted as deleted, and the oldest day in the
|
|
26
|
+
window may be incomplete.
|
|
27
|
+
|
|
28
|
+
### Changed
|
|
29
|
+
- `inspect` and `analyze` raise `TableNotFoundError` when the catalog does not exist.
|
|
30
|
+
Before, every fact in `inspect` was `n/a` and `analyze` raised `RuntimeError` about the
|
|
31
|
+
schema.
|
|
32
|
+
|
|
7
33
|
## [0.2.0] - 2026-10-08
|
|
8
34
|
|
|
9
35
|
Column analysis (`analyze`) and an excerpt of the Delta history in `inspect`.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: dataowl
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Facts about Databricks data products for building dbt staging models.
|
|
5
5
|
Project-URL: Homepage, https://github.com/alexD1990/dataowl
|
|
6
6
|
Project-URL: Repository, https://github.com/alexD1990/dataowl
|
|
@@ -33,8 +33,10 @@ Description-Content-Type: text/markdown
|
|
|
33
33
|
dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
|
|
34
34
|
rows, schema, table properties and an excerpt of the Delta history, and facts about the
|
|
35
35
|
columns you choose: grain and uniqueness of keys, time span and cadence of time columns, and
|
|
36
|
-
how two time columns compare.
|
|
37
|
-
|
|
36
|
+
how two time columns compare. From the Delta history it also reports how the table is
|
|
37
|
+
written: commits per day and per hour of day, operations, and rows inserted, updated and
|
|
38
|
+
deleted per operation. It is meant for data engineers who build a dbt staging model and want
|
|
39
|
+
the facts without writing exploratory SQL. It reports what exists and never gives
|
|
38
40
|
recommendations.
|
|
39
41
|
|
|
40
42
|
## Requirements
|
|
@@ -49,13 +51,17 @@ recommendations.
|
|
|
49
51
|
In a Databricks notebook:
|
|
50
52
|
|
|
51
53
|
```python
|
|
52
|
-
%pip install dataowl==0.
|
|
54
|
+
%pip install dataowl==0.3.0
|
|
53
55
|
```
|
|
54
56
|
|
|
55
57
|
## Usage
|
|
56
58
|
|
|
57
59
|
dataowl works in two steps: `inspect` gives an overview of the table, and `analyze` gives
|
|
58
|
-
facts about columns you choose.
|
|
60
|
+
facts about columns you choose. In addition, `history` gives detailed facts from the Delta
|
|
61
|
+
history about how the table is written.
|
|
62
|
+
|
|
63
|
+
`inspect`, `analyze` and `history` raise `TableNotFoundError` when the table does not exist
|
|
64
|
+
or is not accessible, including when the catalog does not exist.
|
|
59
65
|
|
|
60
66
|
### inspect
|
|
61
67
|
|
|
@@ -158,7 +164,86 @@ analysis.to_dict() # JSON-serializable, like Overview.to_dict()
|
|
|
158
164
|
validated before any query against the table's data.
|
|
159
165
|
- `RuntimeError` when no SparkSession is available, and when the schema cannot be read from
|
|
160
166
|
`information_schema.columns`, since the columns cannot then be validated.
|
|
161
|
-
- `TableNotFoundError` when the table does not exist or is not accessible
|
|
167
|
+
- `TableNotFoundError` when the table does not exist or is not accessible, including when
|
|
168
|
+
the catalog does not exist.
|
|
169
|
+
|
|
170
|
+
Every other failure makes the affected facts `n/a (<reason>)`.
|
|
171
|
+
|
|
172
|
+
### history
|
|
173
|
+
|
|
174
|
+
```python
|
|
175
|
+
dataowl.history("catalog.schema.table").show()
|
|
176
|
+
dataowl.history("catalog.schema.table", limit=100).show() # only the newest 100 commits
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
Full signature:
|
|
180
|
+
|
|
181
|
+
```python
|
|
182
|
+
history(
|
|
183
|
+
table: str,
|
|
184
|
+
*,
|
|
185
|
+
spark: SparkSession | None = None,
|
|
186
|
+
limit: int | None = None,
|
|
187
|
+
) -> HistoryAnalysis
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
`history` reads `DESCRIBE HISTORY` and reports:
|
|
191
|
+
|
|
192
|
+
- **The observation window:** the oldest and newest commit, the number of days and the
|
|
193
|
+
number of commits. The window is shown at the top, because every other number only
|
|
194
|
+
covers this period.
|
|
195
|
+
- **Commits per day:** median, min and max over every calendar day in the window. Days
|
|
196
|
+
without commits count as 0.
|
|
197
|
+
- **Commits per hour of day:** the number of commits in each hour, 0–23.
|
|
198
|
+
- **Operations:** the number of commits per operation, with `WRITE (overwrite)` as its own
|
|
199
|
+
category, as in `inspect`.
|
|
200
|
+
- **Rows per operation:** inserted, updated and deleted rows, from `operationMetrics`.
|
|
201
|
+
|
|
202
|
+
`limit` reads only the newest `limit` commits (`DESCRIBE HISTORY ... LIMIT <limit>`). Without
|
|
203
|
+
`limit`, every commit in the history is read.
|
|
204
|
+
|
|
205
|
+
Commit timestamps are used as Databricks returns them, in the session time zone, without
|
|
206
|
+
conversion. Days and hours of day are counted in that time zone, and its name is shown as
|
|
207
|
+
**Time zone** (from `current_timezone()`). The number of days counts calendar days from the
|
|
208
|
+
date of the oldest commit to the date of the newest, both included.
|
|
209
|
+
|
|
210
|
+
Rows per operation are based on these `operationMetrics` keys:
|
|
211
|
+
|
|
212
|
+
| Operation | inserted | updated | deleted |
|
|
213
|
+
|---|---|---|---|
|
|
214
|
+
| `MERGE` | `numTargetRowsInserted` | `numTargetRowsUpdated` | `numTargetRowsDeleted` |
|
|
215
|
+
| `WRITE`, `WRITE (overwrite)`, operations ending with `AS SELECT` | `numOutputRows` | | |
|
|
216
|
+
| `UPDATE` | | `numUpdatedRows` | |
|
|
217
|
+
| `DELETE` | | | `numDeletedRows` |
|
|
218
|
+
|
|
219
|
+
Other combinations of operation and row type are not shown. **commits** is shown as the
|
|
220
|
+
number of commits that have the metric out of all commits of the operation, for example
|
|
221
|
+
`27/29`. Sum, median and max cover only the commits that have the metric, and are shown as
|
|
222
|
+
`–` when none has it. If a metric has an unexpected format, only the rows per operation are
|
|
223
|
+
`n/a (<reason>)`.
|
|
224
|
+
|
|
225
|
+
The output ends with three notes:
|
|
226
|
+
|
|
227
|
+
- The history is limited by `delta.logRetentionDuration`; the counts cover the window shown.
|
|
228
|
+
- Rows replaced by `WRITE (overwrite)` are not counted as deleted.
|
|
229
|
+
- The oldest day in the window may be incomplete, because `limit` or the log retention can
|
|
230
|
+
cut the history in the middle of a day.
|
|
231
|
+
|
|
232
|
+
```python
|
|
233
|
+
analysis = dataowl.history("catalog.schema.table")
|
|
234
|
+
analysis.to_dict() # JSON-serializable, like Overview.to_dict()
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
For views, every fact except the time zone is `n/a (Not available for views)`, and
|
|
238
|
+
`DESCRIBE HISTORY` is not run.
|
|
239
|
+
|
|
240
|
+
`history` raises:
|
|
241
|
+
|
|
242
|
+
- `ValueError` for an invalid table name, and when `limit` is not `None` or an integer of
|
|
243
|
+
at least 1.
|
|
244
|
+
- `RuntimeError` when no SparkSession is available.
|
|
245
|
+
- `TableNotFoundError` when the table does not exist or is not accessible, including when
|
|
246
|
+
the catalog does not exist.
|
|
162
247
|
|
|
163
248
|
Every other failure makes the affected facts `n/a (<reason>)`.
|
|
164
249
|
|
|
@@ -262,6 +347,33 @@ COMPARE tpep_pickup_datetime → tpep_dropoff_datetime
|
|
|
262
347
|
The data in `samples.nyctaxi.trips` is from January and February 2016, so every day in the
|
|
263
348
|
rows-per-day window has 0 rows.
|
|
264
349
|
|
|
350
|
+
```python
|
|
351
|
+
dataowl.history("samples.nyctaxi.trips").show()
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
```text
|
|
355
|
+
samples.nyctaxi.trips
|
|
356
|
+
|
|
357
|
+
HISTORY (2025-09-09 15:05 – 2026-09-14 15:07, 371 days, 223 commits)
|
|
358
|
+
Time zone: Etc/UTC (session)
|
|
359
|
+
Commits per day: median 1 · min 0 · max 2
|
|
360
|
+
Commits per hour of day:
|
|
361
|
+
06:00 1
|
|
362
|
+
15:00 222
|
|
363
|
+
|
|
364
|
+
OPERATIONS
|
|
365
|
+
CREATE OR REPLACE TABLE AS SELECT: 212
|
|
366
|
+
SET TBLPROPERTIES: 11
|
|
367
|
+
|
|
368
|
+
ROWS
|
|
369
|
+
operation rows commits sum median max
|
|
370
|
+
CREATE OR REPLACE TABLE AS SELECT inserted 212/212 4 649 584 21 932 21 932
|
|
371
|
+
|
|
372
|
+
Note: history is limited by delta.logRetentionDuration; counts cover the window above.
|
|
373
|
+
Note: rows replaced by WRITE (overwrite) are not counted as deleted.
|
|
374
|
+
Note: the oldest day in the window may be incomplete.
|
|
375
|
+
```
|
|
376
|
+
|
|
265
377
|
## Principles
|
|
266
378
|
|
|
267
379
|
- **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
|
|
@@ -272,8 +384,8 @@ rows-per-day window has 0 rows.
|
|
|
272
384
|
- **You choose the columns.** `analyze` only analyzes the columns you give.
|
|
273
385
|
- **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
|
|
274
386
|
object type make the affected facts `n/a (<reason>)`; the rest of the report is still
|
|
275
|
-
produced. Only a table that does not exist or is not accessible
|
|
276
|
-
`TableNotFoundError`.
|
|
387
|
+
produced. Only a table that does not exist or is not accessible, including a table in a
|
|
388
|
+
catalog that does not exist, raises `TableNotFoundError`.
|
|
277
389
|
|
|
278
390
|
## Facts per object type
|
|
279
391
|
|
|
@@ -285,6 +397,7 @@ rows-per-day window has 0 rows.
|
|
|
285
397
|
| Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
|
|
286
398
|
| Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
287
399
|
| History excerpt | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
400
|
+
| `history()`: window, commits, operations, rows per operation | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
288
401
|
| Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
|
|
289
402
|
|
|
290
403
|
¹ The query runs. If Databricks does not support it for the object, the facts are shown as
|
|
@@ -334,6 +447,25 @@ original value is shown in the header. "n/a type" means the object type could no
|
|
|
334
447
|
- Medians and means are rounded to one decimal and shown without decimals when the rounded
|
|
335
448
|
value is whole.
|
|
336
449
|
|
|
450
|
+
### history
|
|
451
|
+
|
|
452
|
+
- **HISTORY (... days, ... commits)** is the oldest and newest commit, the number of
|
|
453
|
+
calendar days from the date of the oldest to the date of the newest commit, both
|
|
454
|
+
included, and the number of commits.
|
|
455
|
+
- **Limit** is shown only when `limit` is given: only the newest `limit` commits are read.
|
|
456
|
+
- **Time zone** is the session time zone, from `current_timezone()`. Commit timestamps,
|
|
457
|
+
days and hours of day are in this time zone.
|
|
458
|
+
- **Commits per day** is the median, min and max over every day in the window, with 0 for
|
|
459
|
+
days without commits.
|
|
460
|
+
- **Commits per hour of day** lists the hours of day that have commits, with the number of
|
|
461
|
+
commits.
|
|
462
|
+
- **OPERATIONS** is the number of commits per operation, as in the HISTORY block of
|
|
463
|
+
`inspect`.
|
|
464
|
+
- **ROWS** has one line per operation and row type (inserted, updated, deleted) with a
|
|
465
|
+
metric, see the table under **history** above. **commits** is the number of commits with
|
|
466
|
+
the metric out of all commits of the operation. **sum**, **median** and **max** cover the
|
|
467
|
+
commits with the metric, and are `–` when none has it.
|
|
468
|
+
|
|
337
469
|
## Known deviations
|
|
338
470
|
|
|
339
471
|
- `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
|
|
@@ -3,8 +3,10 @@
|
|
|
3
3
|
dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
|
|
4
4
|
rows, schema, table properties and an excerpt of the Delta history, and facts about the
|
|
5
5
|
columns you choose: grain and uniqueness of keys, time span and cadence of time columns, and
|
|
6
|
-
how two time columns compare.
|
|
7
|
-
|
|
6
|
+
how two time columns compare. From the Delta history it also reports how the table is
|
|
7
|
+
written: commits per day and per hour of day, operations, and rows inserted, updated and
|
|
8
|
+
deleted per operation. It is meant for data engineers who build a dbt staging model and want
|
|
9
|
+
the facts without writing exploratory SQL. It reports what exists and never gives
|
|
8
10
|
recommendations.
|
|
9
11
|
|
|
10
12
|
## Requirements
|
|
@@ -19,13 +21,17 @@ recommendations.
|
|
|
19
21
|
In a Databricks notebook:
|
|
20
22
|
|
|
21
23
|
```python
|
|
22
|
-
%pip install dataowl==0.
|
|
24
|
+
%pip install dataowl==0.3.0
|
|
23
25
|
```
|
|
24
26
|
|
|
25
27
|
## Usage
|
|
26
28
|
|
|
27
29
|
dataowl works in two steps: `inspect` gives an overview of the table, and `analyze` gives
|
|
28
|
-
facts about columns you choose.
|
|
30
|
+
facts about columns you choose. In addition, `history` gives detailed facts from the Delta
|
|
31
|
+
history about how the table is written.
|
|
32
|
+
|
|
33
|
+
`inspect`, `analyze` and `history` raise `TableNotFoundError` when the table does not exist
|
|
34
|
+
or is not accessible, including when the catalog does not exist.
|
|
29
35
|
|
|
30
36
|
### inspect
|
|
31
37
|
|
|
@@ -128,7 +134,86 @@ analysis.to_dict() # JSON-serializable, like Overview.to_dict()
|
|
|
128
134
|
validated before any query against the table's data.
|
|
129
135
|
- `RuntimeError` when no SparkSession is available, and when the schema cannot be read from
|
|
130
136
|
`information_schema.columns`, since the columns cannot then be validated.
|
|
131
|
-
- `TableNotFoundError` when the table does not exist or is not accessible
|
|
137
|
+
- `TableNotFoundError` when the table does not exist or is not accessible, including when
|
|
138
|
+
the catalog does not exist.
|
|
139
|
+
|
|
140
|
+
Every other failure makes the affected facts `n/a (<reason>)`.
|
|
141
|
+
|
|
142
|
+
### history
|
|
143
|
+
|
|
144
|
+
```python
|
|
145
|
+
dataowl.history("catalog.schema.table").show()
|
|
146
|
+
dataowl.history("catalog.schema.table", limit=100).show() # only the newest 100 commits
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
Full signature:
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
history(
|
|
153
|
+
table: str,
|
|
154
|
+
*,
|
|
155
|
+
spark: SparkSession | None = None,
|
|
156
|
+
limit: int | None = None,
|
|
157
|
+
) -> HistoryAnalysis
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
`history` reads `DESCRIBE HISTORY` and reports:
|
|
161
|
+
|
|
162
|
+
- **The observation window:** the oldest and newest commit, the number of days and the
|
|
163
|
+
number of commits. The window is shown at the top, because every other number only
|
|
164
|
+
covers this period.
|
|
165
|
+
- **Commits per day:** median, min and max over every calendar day in the window. Days
|
|
166
|
+
without commits count as 0.
|
|
167
|
+
- **Commits per hour of day:** the number of commits in each hour, 0–23.
|
|
168
|
+
- **Operations:** the number of commits per operation, with `WRITE (overwrite)` as its own
|
|
169
|
+
category, as in `inspect`.
|
|
170
|
+
- **Rows per operation:** inserted, updated and deleted rows, from `operationMetrics`.
|
|
171
|
+
|
|
172
|
+
`limit` reads only the newest `limit` commits (`DESCRIBE HISTORY ... LIMIT <limit>`). Without
|
|
173
|
+
`limit`, every commit in the history is read.
|
|
174
|
+
|
|
175
|
+
Commit timestamps are used as Databricks returns them, in the session time zone, without
|
|
176
|
+
conversion. Days and hours of day are counted in that time zone, and its name is shown as
|
|
177
|
+
**Time zone** (from `current_timezone()`). The number of days counts calendar days from the
|
|
178
|
+
date of the oldest commit to the date of the newest, both included.
|
|
179
|
+
|
|
180
|
+
Rows per operation are based on these `operationMetrics` keys:
|
|
181
|
+
|
|
182
|
+
| Operation | inserted | updated | deleted |
|
|
183
|
+
|---|---|---|---|
|
|
184
|
+
| `MERGE` | `numTargetRowsInserted` | `numTargetRowsUpdated` | `numTargetRowsDeleted` |
|
|
185
|
+
| `WRITE`, `WRITE (overwrite)`, operations ending with `AS SELECT` | `numOutputRows` | | |
|
|
186
|
+
| `UPDATE` | | `numUpdatedRows` | |
|
|
187
|
+
| `DELETE` | | | `numDeletedRows` |
|
|
188
|
+
|
|
189
|
+
Other combinations of operation and row type are not shown. **commits** is shown as the
|
|
190
|
+
number of commits that have the metric out of all commits of the operation, for example
|
|
191
|
+
`27/29`. Sum, median and max cover only the commits that have the metric, and are shown as
|
|
192
|
+
`–` when none has it. If a metric has an unexpected format, only the rows per operation are
|
|
193
|
+
`n/a (<reason>)`.
|
|
194
|
+
|
|
195
|
+
The output ends with three notes:
|
|
196
|
+
|
|
197
|
+
- The history is limited by `delta.logRetentionDuration`; the counts cover the window shown.
|
|
198
|
+
- Rows replaced by `WRITE (overwrite)` are not counted as deleted.
|
|
199
|
+
- The oldest day in the window may be incomplete, because `limit` or the log retention can
|
|
200
|
+
cut the history in the middle of a day.
|
|
201
|
+
|
|
202
|
+
```python
|
|
203
|
+
analysis = dataowl.history("catalog.schema.table")
|
|
204
|
+
analysis.to_dict() # JSON-serializable, like Overview.to_dict()
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
For views, every fact except the time zone is `n/a (Not available for views)`, and
|
|
208
|
+
`DESCRIBE HISTORY` is not run.
|
|
209
|
+
|
|
210
|
+
`history` raises:
|
|
211
|
+
|
|
212
|
+
- `ValueError` for an invalid table name, and when `limit` is not `None` or an integer of
|
|
213
|
+
at least 1.
|
|
214
|
+
- `RuntimeError` when no SparkSession is available.
|
|
215
|
+
- `TableNotFoundError` when the table does not exist or is not accessible, including when
|
|
216
|
+
the catalog does not exist.
|
|
132
217
|
|
|
133
218
|
Every other failure makes the affected facts `n/a (<reason>)`.
|
|
134
219
|
|
|
@@ -232,6 +317,33 @@ COMPARE tpep_pickup_datetime → tpep_dropoff_datetime
|
|
|
232
317
|
The data in `samples.nyctaxi.trips` is from January and February 2016, so every day in the
|
|
233
318
|
rows-per-day window has 0 rows.
|
|
234
319
|
|
|
320
|
+
```python
|
|
321
|
+
dataowl.history("samples.nyctaxi.trips").show()
|
|
322
|
+
```
|
|
323
|
+
|
|
324
|
+
```text
|
|
325
|
+
samples.nyctaxi.trips
|
|
326
|
+
|
|
327
|
+
HISTORY (2025-09-09 15:05 – 2026-09-14 15:07, 371 days, 223 commits)
|
|
328
|
+
Time zone: Etc/UTC (session)
|
|
329
|
+
Commits per day: median 1 · min 0 · max 2
|
|
330
|
+
Commits per hour of day:
|
|
331
|
+
06:00 1
|
|
332
|
+
15:00 222
|
|
333
|
+
|
|
334
|
+
OPERATIONS
|
|
335
|
+
CREATE OR REPLACE TABLE AS SELECT: 212
|
|
336
|
+
SET TBLPROPERTIES: 11
|
|
337
|
+
|
|
338
|
+
ROWS
|
|
339
|
+
operation rows commits sum median max
|
|
340
|
+
CREATE OR REPLACE TABLE AS SELECT inserted 212/212 4 649 584 21 932 21 932
|
|
341
|
+
|
|
342
|
+
Note: history is limited by delta.logRetentionDuration; counts cover the window above.
|
|
343
|
+
Note: rows replaced by WRITE (overwrite) are not counted as deleted.
|
|
344
|
+
Note: the oldest day in the window may be incomplete.
|
|
345
|
+
```
|
|
346
|
+
|
|
235
347
|
## Principles
|
|
236
348
|
|
|
237
349
|
- **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
|
|
@@ -242,8 +354,8 @@ rows-per-day window has 0 rows.
|
|
|
242
354
|
- **You choose the columns.** `analyze` only analyzes the columns you give.
|
|
243
355
|
- **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
|
|
244
356
|
object type make the affected facts `n/a (<reason>)`; the rest of the report is still
|
|
245
|
-
produced. Only a table that does not exist or is not accessible
|
|
246
|
-
`TableNotFoundError`.
|
|
357
|
+
produced. Only a table that does not exist or is not accessible, including a table in a
|
|
358
|
+
catalog that does not exist, raises `TableNotFoundError`.
|
|
247
359
|
|
|
248
360
|
## Facts per object type
|
|
249
361
|
|
|
@@ -255,6 +367,7 @@ rows-per-day window has 0 rows.
|
|
|
255
367
|
| Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
|
|
256
368
|
| Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
257
369
|
| History excerpt | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
370
|
+
| `history()`: window, commits, operations, rows per operation | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
|
|
258
371
|
| Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
|
|
259
372
|
|
|
260
373
|
¹ The query runs. If Databricks does not support it for the object, the facts are shown as
|
|
@@ -304,6 +417,25 @@ original value is shown in the header. "n/a type" means the object type could no
|
|
|
304
417
|
- Medians and means are rounded to one decimal and shown without decimals when the rounded
|
|
305
418
|
value is whole.
|
|
306
419
|
|
|
420
|
+
### history
|
|
421
|
+
|
|
422
|
+
- **HISTORY (... days, ... commits)** is the oldest and newest commit, the number of
|
|
423
|
+
calendar days from the date of the oldest to the date of the newest commit, both
|
|
424
|
+
included, and the number of commits.
|
|
425
|
+
- **Limit** is shown only when `limit` is given: only the newest `limit` commits are read.
|
|
426
|
+
- **Time zone** is the session time zone, from `current_timezone()`. Commit timestamps,
|
|
427
|
+
days and hours of day are in this time zone.
|
|
428
|
+
- **Commits per day** is the median, min and max over every day in the window, with 0 for
|
|
429
|
+
days without commits.
|
|
430
|
+
- **Commits per hour of day** lists the hours of day that have commits, with the number of
|
|
431
|
+
commits.
|
|
432
|
+
- **OPERATIONS** is the number of commits per operation, as in the HISTORY block of
|
|
433
|
+
`inspect`.
|
|
434
|
+
- **ROWS** has one line per operation and row type (inserted, updated, deleted) with a
|
|
435
|
+
metric, see the table under **history** above. **commits** is the number of commits with
|
|
436
|
+
the metric out of all commits of the operation. **sum**, **median** and **max** cover the
|
|
437
|
+
commits with the metric, and are `–` when none has it.
|
|
438
|
+
|
|
307
439
|
## Known deviations
|
|
308
440
|
|
|
309
441
|
- `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
|
|
@@ -9,7 +9,13 @@ from dataowl.collect.compare import collect_comparison, normalize_compare
|
|
|
9
9
|
from dataowl.collect.constraints import collect_primary_key
|
|
10
10
|
from dataowl.collect.counts import collect_row_count
|
|
11
11
|
from dataowl.collect.detail import collect_detail
|
|
12
|
-
from dataowl.collect.history import
|
|
12
|
+
from dataowl.collect.history import (
|
|
13
|
+
build_history_analysis,
|
|
14
|
+
collect_commits,
|
|
15
|
+
collect_history_excerpt,
|
|
16
|
+
collect_session_time_zone,
|
|
17
|
+
validate_limit,
|
|
18
|
+
)
|
|
13
19
|
from dataowl.collect.keys import KeyInput, collect_key_analyses, normalize_keys
|
|
14
20
|
from dataowl.collect.properties import collect_properties
|
|
15
21
|
from dataowl.collect.tables import collect_table_info
|
|
@@ -21,13 +27,22 @@ from dataowl.collect.timestamps import (
|
|
|
21
27
|
from dataowl.errors import TableNotFoundError
|
|
22
28
|
from dataowl.identifiers import parse_table
|
|
23
29
|
from dataowl.model.analysis import ColumnAnalysis
|
|
30
|
+
from dataowl.model.history import HistoryAnalysis
|
|
24
31
|
from dataowl.model.overview import Overview
|
|
25
32
|
from dataowl.runner import get_runner
|
|
26
33
|
|
|
27
34
|
if TYPE_CHECKING:
|
|
28
35
|
from pyspark.sql import SparkSession
|
|
29
36
|
|
|
30
|
-
__all__ = [
|
|
37
|
+
__all__ = [
|
|
38
|
+
"ColumnAnalysis",
|
|
39
|
+
"HistoryAnalysis",
|
|
40
|
+
"Overview",
|
|
41
|
+
"TableNotFoundError",
|
|
42
|
+
"analyze",
|
|
43
|
+
"history",
|
|
44
|
+
"inspect",
|
|
45
|
+
]
|
|
31
46
|
|
|
32
47
|
|
|
33
48
|
def inspect(
|
|
@@ -36,8 +51,9 @@ def inspect(
|
|
|
36
51
|
"""Collect overview facts for a table.
|
|
37
52
|
|
|
38
53
|
Raises ValueError for an invalid table name, RuntimeError when no SparkSession is
|
|
39
|
-
available, and TableNotFoundError when the table does not exist or is not accessible
|
|
40
|
-
Every other failure makes the affected facts
|
|
54
|
+
available, and TableNotFoundError when the table does not exist or is not accessible,
|
|
55
|
+
including when the catalog does not exist. Every other failure makes the affected facts
|
|
56
|
+
unavailable.
|
|
41
57
|
"""
|
|
42
58
|
ref = parse_table(table)
|
|
43
59
|
runner = get_runner(spark)
|
|
@@ -98,8 +114,8 @@ def analyze(
|
|
|
98
114
|
None, for invalid days, and for an unknown or unsupported column. All input is validated
|
|
99
115
|
before any query against the table's data. Raises RuntimeError when no SparkSession is
|
|
100
116
|
available and when the schema is not available from information_schema.columns. Raises
|
|
101
|
-
TableNotFoundError when the table does not exist or is not accessible
|
|
102
|
-
failure makes the affected facts unavailable.
|
|
117
|
+
TableNotFoundError when the table does not exist or is not accessible, including when
|
|
118
|
+
the catalog does not exist. Every other failure makes the affected facts unavailable.
|
|
103
119
|
"""
|
|
104
120
|
ref = parse_table(table)
|
|
105
121
|
if key is None and timestamp is None and compare is None:
|
|
@@ -125,3 +141,27 @@ def analyze(
|
|
|
125
141
|
timestamps=collect_timestamp_analyses(runner, ref, time_columns, days),
|
|
126
142
|
comparison=collect_comparison(runner, ref, *pair) if pair is not None else None,
|
|
127
143
|
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def history(
|
|
147
|
+
table: str, *, spark: SparkSession | None = None, limit: int | None = None
|
|
148
|
+
) -> HistoryAnalysis:
|
|
149
|
+
"""Collect facts about how the table is written, from DESCRIBE HISTORY.
|
|
150
|
+
|
|
151
|
+
With limit, only the newest `limit` commits are read. Timestamps are used in the session
|
|
152
|
+
time zone, which is collected as a fact.
|
|
153
|
+
|
|
154
|
+
Raises ValueError for an invalid table name and when limit is not None or an int of at
|
|
155
|
+
least 1, RuntimeError when no SparkSession is available, and TableNotFoundError when the
|
|
156
|
+
table does not exist or is not accessible, including when the catalog does not exist.
|
|
157
|
+
Every other failure makes the affected facts unavailable; for a view, the history facts
|
|
158
|
+
are unavailable.
|
|
159
|
+
"""
|
|
160
|
+
ref = parse_table(table)
|
|
161
|
+
validate_limit(limit)
|
|
162
|
+
runner = get_runner(spark)
|
|
163
|
+
|
|
164
|
+
info = collect_table_info(runner, ref)
|
|
165
|
+
time_zone = collect_session_time_zone(runner)
|
|
166
|
+
result = collect_commits(runner, ref, info.object_type, limit)
|
|
167
|
+
return build_history_analysis(ref, limit, time_zone, result.commits, result.metrics_reason)
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Collection layer: queries to raw data to model objects. The only layer that knows SQL."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from dataowl.model.facts import Fact
|
|
9
|
+
from dataowl.model.overview import ObjectType
|
|
10
|
+
from dataowl.runner import SqlRunner
|
|
11
|
+
|
|
12
|
+
_MAX_REASON_LENGTH = 200
|
|
13
|
+
|
|
14
|
+
NOT_FOR_VIEWS = "Not available for views"
|
|
15
|
+
|
|
16
|
+
_CATALOG_NOT_FOUND_CLASSES = frozenset({"NO_SUCH_CATALOG_EXCEPTION", "CATALOG_NOT_FOUND"})
|
|
17
|
+
_TABLE_OR_VIEW_NOT_FOUND = "TABLE_OR_VIEW_NOT_FOUND"
|
|
18
|
+
_LEADING_ERROR_CLASS = re.compile(r"\[([A-Za-z0-9_.]+)\]")
|
|
19
|
+
_INFORMATION_SCHEMA = re.compile(r"(?<!\w)information_schema(?!\w)", re.IGNORECASE)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def is_view(object_type: Fact[ObjectType]) -> bool:
|
|
23
|
+
"""True only when the object type is known to be VIEW."""
|
|
24
|
+
return object_type.available and object_type.value is ObjectType.VIEW
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def safe_query(
|
|
28
|
+
runner: SqlRunner, sql: str, params: dict[str, Any] | None = None
|
|
29
|
+
) -> list[dict[str, Any]] | Exception:
|
|
30
|
+
"""Run a query and return the rows, or the exception instead of raising it."""
|
|
31
|
+
try:
|
|
32
|
+
return runner.query(sql, params)
|
|
33
|
+
except Exception as exc:
|
|
34
|
+
return exc
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def is_catalog_not_found(exc: Exception) -> bool:
|
|
38
|
+
"""True when the error says that the catalog does not exist.
|
|
39
|
+
|
|
40
|
+
The error class is read from getCondition(), then getErrorClass(), and otherwise from a
|
|
41
|
+
leading "[CLASS]" in the first line of the message. PySpark is not imported. True for
|
|
42
|
+
NO_SUCH_CATALOG_EXCEPTION and CATALOG_NOT_FOUND, and for TABLE_OR_VIEW_NOT_FOUND when the
|
|
43
|
+
first line names information_schema, quoted or not. False for everything else, including
|
|
44
|
+
access errors.
|
|
45
|
+
"""
|
|
46
|
+
lines = str(exc).strip().splitlines()
|
|
47
|
+
first_line = lines[0].strip() if lines else ""
|
|
48
|
+
error_class = _error_class(exc, first_line)
|
|
49
|
+
if error_class in _CATALOG_NOT_FOUND_CLASSES:
|
|
50
|
+
return True
|
|
51
|
+
if error_class == _TABLE_OR_VIEW_NOT_FOUND:
|
|
52
|
+
return _INFORMATION_SCHEMA.search(first_line) is not None
|
|
53
|
+
return False
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _error_class(exc: Exception, first_line: str) -> str | None:
|
|
57
|
+
for method_name in ("getCondition", "getErrorClass"):
|
|
58
|
+
method = getattr(exc, method_name, None)
|
|
59
|
+
if not callable(method):
|
|
60
|
+
continue
|
|
61
|
+
try:
|
|
62
|
+
value = method()
|
|
63
|
+
except Exception:
|
|
64
|
+
continue
|
|
65
|
+
if isinstance(value, str) and value.strip():
|
|
66
|
+
return value.strip().upper()
|
|
67
|
+
match = _LEADING_ERROR_CLASS.match(first_line)
|
|
68
|
+
return match.group(1).upper() if match else None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def short_reason(exc: Exception) -> str:
|
|
72
|
+
"""First line of the error message, at most 200 characters.
|
|
73
|
+
|
|
74
|
+
Falls back to the exception type name when the message is empty.
|
|
75
|
+
"""
|
|
76
|
+
lines = str(exc).strip().splitlines()
|
|
77
|
+
reason = lines[0].strip() if lines else type(exc).__name__
|
|
78
|
+
if len(reason) > _MAX_REASON_LENGTH:
|
|
79
|
+
reason = reason[: _MAX_REASON_LENGTH - 1] + "…"
|
|
80
|
+
return reason
|