dataowl 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. {dataowl-0.2.0 → dataowl-0.3.0}/CHANGELOG.md +26 -0
  2. {dataowl-0.2.0 → dataowl-0.3.0}/PKG-INFO +140 -8
  3. {dataowl-0.2.0 → dataowl-0.3.0}/README.md +139 -7
  4. {dataowl-0.2.0 → dataowl-0.3.0}/pyproject.toml +1 -1
  5. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/__init__.py +46 -6
  6. dataowl-0.3.0/src/dataowl/collect/__init__.py +80 -0
  7. dataowl-0.3.0/src/dataowl/collect/history.py +424 -0
  8. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/tables.py +8 -3
  9. dataowl-0.3.0/src/dataowl/model/history.py +126 -0
  10. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/render/terminal.py +108 -0
  11. dataowl-0.3.0/tests/fixtures/describe_history_metrics.json +83 -0
  12. dataowl-0.3.0/tests/fixtures/history_full.txt +31 -0
  13. dataowl-0.3.0/tests/fixtures/history_limit.txt +23 -0
  14. dataowl-0.3.0/tests/fixtures/history_partial.txt +23 -0
  15. dataowl-0.3.0/tests/fixtures/history_view.txt +8 -0
  16. {dataowl-0.2.0 → dataowl-0.3.0}/tests/integration/setup_test_tables.sql +20 -0
  17. {dataowl-0.2.0 → dataowl-0.3.0}/tests/integration/test_integration.py +134 -2
  18. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_analyze.py +14 -0
  19. dataowl-0.3.0/tests/unit/test_collect.py +147 -0
  20. dataowl-0.3.0/tests/unit/test_collect_history.py +999 -0
  21. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_tables.py +35 -0
  22. dataowl-0.3.0/tests/unit/test_history.py +232 -0
  23. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_inspect.py +16 -0
  24. dataowl-0.3.0/tests/unit/test_terminal_history.py +147 -0
  25. dataowl-0.2.0/src/dataowl/collect/__init__.py +0 -40
  26. dataowl-0.2.0/src/dataowl/collect/history.py +0 -124
  27. dataowl-0.2.0/src/dataowl/model/history.py +0 -19
  28. dataowl-0.2.0/tests/unit/test_collect.py +0 -44
  29. dataowl-0.2.0/tests/unit/test_collect_history.py +0 -338
  30. {dataowl-0.2.0 → dataowl-0.3.0}/.gitignore +0 -0
  31. {dataowl-0.2.0 → dataowl-0.3.0}/LICENSE +0 -0
  32. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/columns.py +0 -0
  33. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/compare.py +0 -0
  34. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/constraints.py +0 -0
  35. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/counts.py +0 -0
  36. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/detail.py +0 -0
  37. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/keys.py +0 -0
  38. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/properties.py +0 -0
  39. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/collect/timestamps.py +0 -0
  40. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/errors.py +0 -0
  41. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/identifiers.py +0 -0
  42. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/__init__.py +0 -0
  43. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/analysis.py +0 -0
  44. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/facts.py +0 -0
  45. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/model/overview.py +0 -0
  46. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/py.typed +0 -0
  47. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/render/__init__.py +0 -0
  48. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/render/format.py +0 -0
  49. {dataowl-0.2.0 → dataowl-0.3.0}/src/dataowl/runner.py +0 -0
  50. {dataowl-0.2.0 → dataowl-0.3.0}/tests/conftest.py +0 -0
  51. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_days.txt +0 -0
  52. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_full.txt +0 -0
  53. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_minimal.txt +0 -0
  54. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/analysis_unavailable.txt +0 -0
  55. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/describe_detail_table.json +0 -0
  56. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/describe_history_table.json +0 -0
  57. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/overview_table.txt +0 -0
  58. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/overview_unavailable.txt +0 -0
  59. {dataowl-0.2.0 → dataowl-0.3.0}/tests/fixtures/overview_view.txt +0 -0
  60. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_columns.py +0 -0
  61. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_compare.py +0 -0
  62. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_constraints.py +0 -0
  63. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_counts.py +0 -0
  64. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_detail.py +0 -0
  65. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_keys.py +0 -0
  66. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_properties.py +0 -0
  67. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_collect_timestamps.py +0 -0
  68. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_facts.py +0 -0
  69. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_fake_runner.py +0 -0
  70. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_format.py +0 -0
  71. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_identifiers.py +0 -0
  72. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_import.py +0 -0
  73. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_import_boundaries.py +0 -0
  74. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_no_judgement.py +0 -0
  75. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_read_only.py +0 -0
  76. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_runner.py +0 -0
  77. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_terminal.py +0 -0
  78. {dataowl-0.2.0 → dataowl-0.3.0}/tests/unit/test_terminal_analysis.py +0 -0
@@ -4,6 +4,32 @@ All notable changes to this project are documented in this file.
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [0.3.0] - 2026-10-08
8
+
9
+ Write behaviour from the Delta history (`history`).
10
+
11
+ ### Added
12
+ - `dataowl.history(table, *, spark=None, limit=None)` returning a `HistoryAnalysis` with
13
+ `to_dict()` and `show()`. `limit` reads only the newest `limit` commits.
14
+ - Observation window (oldest and newest commit, calendar days, commits), commits per day
15
+ (median, min, max, with 0 for days without commits), commits per hour of day, and
16
+ commits per operation with `WRITE (overwrite)` as its own category.
17
+ - Inserted, updated and deleted rows per operation from `operationMetrics`
18
+ (`numTargetRowsInserted`, `numTargetRowsUpdated` and `numTargetRowsDeleted` for `MERGE`,
19
+ `numOutputRows` for `WRITE`, `WRITE (overwrite)` and `... AS SELECT`, `numUpdatedRows`
20
+ for `UPDATE`, `numDeletedRows` for `DELETE`): commits with the metric out of all, sum,
21
+ median and max.
22
+ - The session time zone (`current_timezone()`) as a fact. Days and hours are counted in
23
+ that time zone.
24
+ - Notes in the output: the history is limited by `delta.logRetentionDuration`, rows
25
+ replaced by `WRITE (overwrite)` are not counted as deleted, and the oldest day in the
26
+ window may be incomplete.
27
+
28
+ ### Changed
29
+ - `inspect` and `analyze` raise `TableNotFoundError` when the catalog does not exist.
30
+ Before, every fact in `inspect` was `n/a` and `analyze` raised `RuntimeError` about the
31
+ schema.
32
+
7
33
  ## [0.2.0] - 2026-10-08
8
34
 
9
35
  Column analysis (`analyze`) and an excerpt of the Delta history in `inspect`.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: dataowl
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Facts about Databricks data products for building dbt staging models.
5
5
  Project-URL: Homepage, https://github.com/alexD1990/dataowl
6
6
  Project-URL: Repository, https://github.com/alexD1990/dataowl
@@ -33,8 +33,10 @@ Description-Content-Type: text/markdown
33
33
  dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
34
34
  rows, schema, table properties and an excerpt of the Delta history, and facts about the
35
35
  columns you choose: grain and uniqueness of keys, time span and cadence of time columns, and
36
- how two time columns compare. It is meant for data engineers who build a dbt staging model
37
- and want the facts without writing exploratory SQL. It reports what exists and never gives
36
+ how two time columns compare. From the Delta history it also reports how the table is
37
+ written: commits per day and per hour of day, operations, and rows inserted, updated and
38
+ deleted per operation. It is meant for data engineers who build a dbt staging model and want
39
+ the facts without writing exploratory SQL. It reports what exists and never gives
38
40
  recommendations.
39
41
 
40
42
  ## Requirements
@@ -49,13 +51,17 @@ recommendations.
49
51
  In a Databricks notebook:
50
52
 
51
53
  ```python
52
- %pip install dataowl==0.2.0
54
+ %pip install dataowl==0.3.0
53
55
  ```
54
56
 
55
57
  ## Usage
56
58
 
57
59
  dataowl works in two steps: `inspect` gives an overview of the table, and `analyze` gives
58
- facts about columns you choose.
60
+ facts about columns you choose. In addition, `history` gives detailed facts from the Delta
61
+ history about how the table is written.
62
+
63
+ `inspect`, `analyze` and `history` raise `TableNotFoundError` when the table does not exist
64
+ or is not accessible, including when the catalog does not exist.
59
65
 
60
66
  ### inspect
61
67
 
@@ -158,7 +164,86 @@ analysis.to_dict() # JSON-serializable, like Overview.to_dict()
158
164
  validated before any query against the table's data.
159
165
  - `RuntimeError` when no SparkSession is available, and when the schema cannot be read from
160
166
  `information_schema.columns`, since the columns cannot then be validated.
161
- - `TableNotFoundError` when the table does not exist or is not accessible.
167
+ - `TableNotFoundError` when the table does not exist or is not accessible, including when
168
+ the catalog does not exist.
169
+
170
+ Every other failure makes the affected facts `n/a (<reason>)`.
171
+
172
+ ### history
173
+
174
+ ```python
175
+ dataowl.history("catalog.schema.table").show()
176
+ dataowl.history("catalog.schema.table", limit=100).show() # only the newest 100 commits
177
+ ```
178
+
179
+ Full signature:
180
+
181
+ ```python
182
+ history(
183
+ table: str,
184
+ *,
185
+ spark: SparkSession | None = None,
186
+ limit: int | None = None,
187
+ ) -> HistoryAnalysis
188
+ ```
189
+
190
+ `history` reads `DESCRIBE HISTORY` and reports:
191
+
192
+ - **The observation window:** the oldest and newest commit, the number of days and the
193
+ number of commits. The window is shown at the top, because every other number only
194
+ covers this period.
195
+ - **Commits per day:** median, min and max over every calendar day in the window. Days
196
+ without commits count as 0.
197
+ - **Commits per hour of day:** the number of commits in each hour, 0–23.
198
+ - **Operations:** the number of commits per operation, with `WRITE (overwrite)` as its own
199
+ category, as in `inspect`.
200
+ - **Rows per operation:** inserted, updated and deleted rows, from `operationMetrics`.
201
+
202
+ `limit` reads only the newest `limit` commits (`DESCRIBE HISTORY ... LIMIT <limit>`). Without
203
+ `limit`, every commit in the history is read.
204
+
205
+ Commit timestamps are used as Databricks returns them, in the session time zone, without
206
+ conversion. Days and hours of day are counted in that time zone, and its name is shown as
207
+ **Time zone** (from `current_timezone()`). The number of days counts calendar days from the
208
+ date of the oldest commit to the date of the newest, both included.
209
+
210
+ Rows per operation are based on these `operationMetrics` keys:
211
+
212
+ | Operation | inserted | updated | deleted |
213
+ |---|---|---|---|
214
+ | `MERGE` | `numTargetRowsInserted` | `numTargetRowsUpdated` | `numTargetRowsDeleted` |
215
+ | `WRITE`, `WRITE (overwrite)`, operations ending with `AS SELECT` | `numOutputRows` | | |
216
+ | `UPDATE` | | `numUpdatedRows` | |
217
+ | `DELETE` | | | `numDeletedRows` |
218
+
219
+ Other combinations of operation and row type are not shown. **commits** is shown as the
220
+ number of commits that have the metric out of all commits of the operation, for example
221
+ `27/29`. Sum, median and max cover only the commits that have the metric, and are shown as
222
+ `–` when none has it. If a metric has an unexpected format, only the rows per operation are
223
+ `n/a (<reason>)`.
224
+
225
+ The output ends with three notes:
226
+
227
+ - The history is limited by `delta.logRetentionDuration`; the counts cover the window shown.
228
+ - Rows replaced by `WRITE (overwrite)` are not counted as deleted.
229
+ - The oldest day in the window may be incomplete, because `limit` or the log retention can
230
+ cut the history in the middle of a day.
231
+
232
+ ```python
233
+ analysis = dataowl.history("catalog.schema.table")
234
+ analysis.to_dict() # JSON-serializable, like Overview.to_dict()
235
+ ```
236
+
237
+ For views, every fact except the time zone is `n/a (Not available for views)`, and
238
+ `DESCRIBE HISTORY` is not run.
239
+
240
+ `history` raises:
241
+
242
+ - `ValueError` for an invalid table name, and when `limit` is not `None` or an integer of
243
+ at least 1.
244
+ - `RuntimeError` when no SparkSession is available.
245
+ - `TableNotFoundError` when the table does not exist or is not accessible, including when
246
+ the catalog does not exist.
162
247
 
163
248
  Every other failure makes the affected facts `n/a (<reason>)`.
164
249
 
@@ -262,6 +347,33 @@ COMPARE tpep_pickup_datetime → tpep_dropoff_datetime
262
347
  The data in `samples.nyctaxi.trips` is from January and February 2016, so every day in the
263
348
  rows-per-day window has 0 rows.
264
349
 
350
+ ```python
351
+ dataowl.history("samples.nyctaxi.trips").show()
352
+ ```
353
+
354
+ ```text
355
+ samples.nyctaxi.trips
356
+
357
+ HISTORY (2025-09-09 15:05 – 2026-09-14 15:07, 371 days, 223 commits)
358
+ Time zone: Etc/UTC (session)
359
+ Commits per day: median 1 · min 0 · max 2
360
+ Commits per hour of day:
361
+ 06:00 1
362
+ 15:00 222
363
+
364
+ OPERATIONS
365
+ CREATE OR REPLACE TABLE AS SELECT: 212
366
+ SET TBLPROPERTIES: 11
367
+
368
+ ROWS
369
+ operation rows commits sum median max
370
+ CREATE OR REPLACE TABLE AS SELECT inserted 212/212 4 649 584 21 932 21 932
371
+
372
+ Note: history is limited by delta.logRetentionDuration; counts cover the window above.
373
+ Note: rows replaced by WRITE (overwrite) are not counted as deleted.
374
+ Note: the oldest day in the window may be incomplete.
375
+ ```
376
+
265
377
  ## Principles
266
378
 
267
379
  - **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
@@ -272,8 +384,8 @@ rows-per-day window has 0 rows.
272
384
  - **You choose the columns.** `analyze` only analyzes the columns you give.
273
385
  - **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
274
386
  object type make the affected facts `n/a (<reason>)`; the rest of the report is still
275
- produced. Only a table that does not exist or is not accessible raises
276
- `TableNotFoundError`.
387
+ produced. Only a table that does not exist or is not accessible, including a table in a
388
+ catalog that does not exist, raises `TableNotFoundError`.
277
389
 
278
390
  ## Facts per object type
279
391
 
@@ -285,6 +397,7 @@ rows-per-day window has 0 rows.
285
397
  | Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
286
398
  | Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
287
399
  | History excerpt | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
400
+ | `history()`: window, commits, operations, rows per operation | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
288
401
  | Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
289
402
 
290
403
  ¹ The query runs. If Databricks does not support it for the object, the facts are shown as
@@ -334,6 +447,25 @@ original value is shown in the header. "n/a type" means the object type could no
334
447
  - Medians and means are rounded to one decimal and shown without decimals when the rounded
335
448
  value is whole.
336
449
 
450
+ ### history
451
+
452
+ - **HISTORY (... days, ... commits)** is the oldest and newest commit, the number of
453
+ calendar days from the date of the oldest to the date of the newest commit, both
454
+ included, and the number of commits.
455
+ - **Limit** is shown only when `limit` is given: only the newest `limit` commits are read.
456
+ - **Time zone** is the session time zone, from `current_timezone()`. Commit timestamps,
457
+ days and hours of day are in this time zone.
458
+ - **Commits per day** is the median, min and max over every day in the window, with 0 for
459
+ days without commits.
460
+ - **Commits per hour of day** lists the hours of day that have commits, with the number of
461
+ commits.
462
+ - **OPERATIONS** is the number of commits per operation, as in the HISTORY block of
463
+ `inspect`.
464
+ - **ROWS** has one line per operation and row type (inserted, updated, deleted) with a
465
+ metric, see the table under **history** above. **commits** is the number of commits with
466
+ the metric out of all commits of the operation. **sum**, **median** and **max** cover the
467
+ commits with the metric, and are `–` when none has it.
468
+
337
469
  ## Known deviations
338
470
 
339
471
  - `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
@@ -3,8 +3,10 @@
3
3
  dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
4
4
  rows, schema, table properties and an excerpt of the Delta history, and facts about the
5
5
  columns you choose: grain and uniqueness of keys, time span and cadence of time columns, and
6
- how two time columns compare. It is meant for data engineers who build a dbt staging model
7
- and want the facts without writing exploratory SQL. It reports what exists and never gives
6
+ how two time columns compare. From the Delta history it also reports how the table is
7
+ written: commits per day and per hour of day, operations, and rows inserted, updated and
8
+ deleted per operation. It is meant for data engineers who build a dbt staging model and want
9
+ the facts without writing exploratory SQL. It reports what exists and never gives
8
10
  recommendations.
9
11
 
10
12
  ## Requirements
@@ -19,13 +21,17 @@ recommendations.
19
21
  In a Databricks notebook:
20
22
 
21
23
  ```python
22
- %pip install dataowl==0.2.0
24
+ %pip install dataowl==0.3.0
23
25
  ```
24
26
 
25
27
  ## Usage
26
28
 
27
29
  dataowl works in two steps: `inspect` gives an overview of the table, and `analyze` gives
28
- facts about columns you choose.
30
+ facts about columns you choose. In addition, `history` gives detailed facts from the Delta
31
+ history about how the table is written.
32
+
33
+ `inspect`, `analyze` and `history` raise `TableNotFoundError` when the table does not exist
34
+ or is not accessible, including when the catalog does not exist.
29
35
 
30
36
  ### inspect
31
37
 
@@ -128,7 +134,86 @@ analysis.to_dict() # JSON-serializable, like Overview.to_dict()
128
134
  validated before any query against the table's data.
129
135
  - `RuntimeError` when no SparkSession is available, and when the schema cannot be read from
130
136
  `information_schema.columns`, since the columns cannot then be validated.
131
- - `TableNotFoundError` when the table does not exist or is not accessible.
137
+ - `TableNotFoundError` when the table does not exist or is not accessible, including when
138
+ the catalog does not exist.
139
+
140
+ Every other failure makes the affected facts `n/a (<reason>)`.
141
+
142
+ ### history
143
+
144
+ ```python
145
+ dataowl.history("catalog.schema.table").show()
146
+ dataowl.history("catalog.schema.table", limit=100).show() # only the newest 100 commits
147
+ ```
148
+
149
+ Full signature:
150
+
151
+ ```python
152
+ history(
153
+ table: str,
154
+ *,
155
+ spark: SparkSession | None = None,
156
+ limit: int | None = None,
157
+ ) -> HistoryAnalysis
158
+ ```
159
+
160
+ `history` reads `DESCRIBE HISTORY` and reports:
161
+
162
+ - **The observation window:** the oldest and newest commit, the number of days and the
163
+ number of commits. The window is shown at the top, because every other number only
164
+ covers this period.
165
+ - **Commits per day:** median, min and max over every calendar day in the window. Days
166
+ without commits count as 0.
167
+ - **Commits per hour of day:** the number of commits in each hour, 0–23.
168
+ - **Operations:** the number of commits per operation, with `WRITE (overwrite)` as its own
169
+ category, as in `inspect`.
170
+ - **Rows per operation:** inserted, updated and deleted rows, from `operationMetrics`.
171
+
172
+ `limit` reads only the newest `limit` commits (`DESCRIBE HISTORY ... LIMIT <limit>`). Without
173
+ `limit`, every commit in the history is read.
174
+
175
+ Commit timestamps are used as Databricks returns them, in the session time zone, without
176
+ conversion. Days and hours of day are counted in that time zone, and its name is shown as
177
+ **Time zone** (from `current_timezone()`). The number of days counts calendar days from the
178
+ date of the oldest commit to the date of the newest, both included.
179
+
180
+ Rows per operation are based on these `operationMetrics` keys:
181
+
182
+ | Operation | inserted | updated | deleted |
183
+ |---|---|---|---|
184
+ | `MERGE` | `numTargetRowsInserted` | `numTargetRowsUpdated` | `numTargetRowsDeleted` |
185
+ | `WRITE`, `WRITE (overwrite)`, operations ending with `AS SELECT` | `numOutputRows` | | |
186
+ | `UPDATE` | | `numUpdatedRows` | |
187
+ | `DELETE` | | | `numDeletedRows` |
188
+
189
+ Other combinations of operation and row type are not shown. **commits** is shown as the
190
+ number of commits that have the metric out of all commits of the operation, for example
191
+ `27/29`. Sum, median and max cover only the commits that have the metric, and are shown as
192
+ `–` when none has it. If a metric has an unexpected format, only the rows per operation are
193
+ `n/a (<reason>)`.
194
+
195
+ The output ends with three notes:
196
+
197
+ - The history is limited by `delta.logRetentionDuration`; the counts cover the window shown.
198
+ - Rows replaced by `WRITE (overwrite)` are not counted as deleted.
199
+ - The oldest day in the window may be incomplete, because `limit` or the log retention can
200
+ cut the history in the middle of a day.
201
+
202
+ ```python
203
+ analysis = dataowl.history("catalog.schema.table")
204
+ analysis.to_dict() # JSON-serializable, like Overview.to_dict()
205
+ ```
206
+
207
+ For views, every fact except the time zone is `n/a (Not available for views)`, and
208
+ `DESCRIBE HISTORY` is not run.
209
+
210
+ `history` raises:
211
+
212
+ - `ValueError` for an invalid table name, and when `limit` is not `None` or an integer of
213
+ at least 1.
214
+ - `RuntimeError` when no SparkSession is available.
215
+ - `TableNotFoundError` when the table does not exist or is not accessible, including when
216
+ the catalog does not exist.
132
217
 
133
218
  Every other failure makes the affected facts `n/a (<reason>)`.
134
219
 
@@ -232,6 +317,33 @@ COMPARE tpep_pickup_datetime → tpep_dropoff_datetime
232
317
  The data in `samples.nyctaxi.trips` is from January and February 2016, so every day in the
233
318
  rows-per-day window has 0 rows.
234
319
 
320
+ ```python
321
+ dataowl.history("samples.nyctaxi.trips").show()
322
+ ```
323
+
324
+ ```text
325
+ samples.nyctaxi.trips
326
+
327
+ HISTORY (2025-09-09 15:05 – 2026-09-14 15:07, 371 days, 223 commits)
328
+ Time zone: Etc/UTC (session)
329
+ Commits per day: median 1 · min 0 · max 2
330
+ Commits per hour of day:
331
+ 06:00 1
332
+ 15:00 222
333
+
334
+ OPERATIONS
335
+ CREATE OR REPLACE TABLE AS SELECT: 212
336
+ SET TBLPROPERTIES: 11
337
+
338
+ ROWS
339
+ operation rows commits sum median max
340
+ CREATE OR REPLACE TABLE AS SELECT inserted 212/212 4 649 584 21 932 21 932
341
+
342
+ Note: history is limited by delta.logRetentionDuration; counts cover the window above.
343
+ Note: rows replaced by WRITE (overwrite) are not counted as deleted.
344
+ Note: the oldest day in the window may be incomplete.
345
+ ```
346
+
235
347
  ## Principles
236
348
 
237
349
  - **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
@@ -242,8 +354,8 @@ rows-per-day window has 0 rows.
242
354
  - **You choose the columns.** `analyze` only analyzes the columns you give.
243
355
  - **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
244
356
  object type make the affected facts `n/a (<reason>)`; the rest of the report is still
245
- produced. Only a table that does not exist or is not accessible raises
246
- `TableNotFoundError`.
357
+ produced. Only a table that does not exist or is not accessible, including a table in a
358
+ catalog that does not exist, raises `TableNotFoundError`.
247
359
 
248
360
  ## Facts per object type
249
361
 
@@ -255,6 +367,7 @@ rows-per-day window has 0 rows.
255
367
  | Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
256
368
  | Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
257
369
  | History excerpt | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
370
+ | `history()`: window, commits, operations, rows per operation | `DESCRIBE HISTORY` | yes | attempted¹ | attempted¹ | no | attempted¹ |
258
371
  | Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
259
372
 
260
373
  ¹ The query runs. If Databricks does not support it for the object, the facts are shown as
@@ -304,6 +417,25 @@ original value is shown in the header. "n/a type" means the object type could no
304
417
  - Medians and means are rounded to one decimal and shown without decimals when the rounded
305
418
  value is whole.
306
419
 
420
+ ### history
421
+
422
+ - **HISTORY (... days, ... commits)** is the oldest and newest commit, the number of
423
+ calendar days from the date of the oldest to the date of the newest commit, both
424
+ included, and the number of commits.
425
+ - **Limit** is shown only when `limit` is given: only the newest `limit` commits are read.
426
+ - **Time zone** is the session time zone, from `current_timezone()`. Commit timestamps,
427
+ days and hours of day are in this time zone.
428
+ - **Commits per day** is the median, min and max over every day in the window, with 0 for
429
+ days without commits.
430
+ - **Commits per hour of day** lists the hours of day that have commits, with the number of
431
+ commits.
432
+ - **OPERATIONS** is the number of commits per operation, as in the HISTORY block of
433
+ `inspect`.
434
+ - **ROWS** has one line per operation and row type (inserted, updated, deleted) with a
435
+ metric, see the table under **history** above. **commits** is the number of commits with
436
+ the metric out of all commits of the operation. **sum**, **median** and **max** cover the
437
+ commits with the metric, and are `–` when none has it.
438
+
307
439
  ## Known deviations
308
440
 
309
441
  - `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "dataowl"
7
- version = "0.2.0"
7
+ version = "0.3.0"
8
8
  description = "Facts about Databricks data products for building dbt staging models."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -9,7 +9,13 @@ from dataowl.collect.compare import collect_comparison, normalize_compare
9
9
  from dataowl.collect.constraints import collect_primary_key
10
10
  from dataowl.collect.counts import collect_row_count
11
11
  from dataowl.collect.detail import collect_detail
12
- from dataowl.collect.history import collect_history_excerpt
12
+ from dataowl.collect.history import (
13
+ build_history_analysis,
14
+ collect_commits,
15
+ collect_history_excerpt,
16
+ collect_session_time_zone,
17
+ validate_limit,
18
+ )
13
19
  from dataowl.collect.keys import KeyInput, collect_key_analyses, normalize_keys
14
20
  from dataowl.collect.properties import collect_properties
15
21
  from dataowl.collect.tables import collect_table_info
@@ -21,13 +27,22 @@ from dataowl.collect.timestamps import (
21
27
  from dataowl.errors import TableNotFoundError
22
28
  from dataowl.identifiers import parse_table
23
29
  from dataowl.model.analysis import ColumnAnalysis
30
+ from dataowl.model.history import HistoryAnalysis
24
31
  from dataowl.model.overview import Overview
25
32
  from dataowl.runner import get_runner
26
33
 
27
34
  if TYPE_CHECKING:
28
35
  from pyspark.sql import SparkSession
29
36
 
30
- __all__ = ["ColumnAnalysis", "Overview", "TableNotFoundError", "analyze", "inspect"]
37
+ __all__ = [
38
+ "ColumnAnalysis",
39
+ "HistoryAnalysis",
40
+ "Overview",
41
+ "TableNotFoundError",
42
+ "analyze",
43
+ "history",
44
+ "inspect",
45
+ ]
31
46
 
32
47
 
33
48
  def inspect(
@@ -36,8 +51,9 @@ def inspect(
36
51
  """Collect overview facts for a table.
37
52
 
38
53
  Raises ValueError for an invalid table name, RuntimeError when no SparkSession is
39
- available, and TableNotFoundError when the table does not exist or is not accessible.
40
- Every other failure makes the affected facts unavailable.
54
+ available, and TableNotFoundError when the table does not exist or is not accessible,
55
+ including when the catalog does not exist. Every other failure makes the affected facts
56
+ unavailable.
41
57
  """
42
58
  ref = parse_table(table)
43
59
  runner = get_runner(spark)
@@ -98,8 +114,8 @@ def analyze(
98
114
  None, for invalid days, and for an unknown or unsupported column. All input is validated
99
115
  before any query against the table's data. Raises RuntimeError when no SparkSession is
100
116
  available and when the schema is not available from information_schema.columns. Raises
101
- TableNotFoundError when the table does not exist or is not accessible. Every other
102
- failure makes the affected facts unavailable.
117
+ TableNotFoundError when the table does not exist or is not accessible, including when
118
+ the catalog does not exist. Every other failure makes the affected facts unavailable.
103
119
  """
104
120
  ref = parse_table(table)
105
121
  if key is None and timestamp is None and compare is None:
@@ -125,3 +141,27 @@ def analyze(
125
141
  timestamps=collect_timestamp_analyses(runner, ref, time_columns, days),
126
142
  comparison=collect_comparison(runner, ref, *pair) if pair is not None else None,
127
143
  )
144
+
145
+
146
+ def history(
147
+ table: str, *, spark: SparkSession | None = None, limit: int | None = None
148
+ ) -> HistoryAnalysis:
149
+ """Collect facts about how the table is written, from DESCRIBE HISTORY.
150
+
151
+ With limit, only the newest `limit` commits are read. Timestamps are used in the session
152
+ time zone, which is collected as a fact.
153
+
154
+ Raises ValueError for an invalid table name and when limit is not None or an int of at
155
+ least 1, RuntimeError when no SparkSession is available, and TableNotFoundError when the
156
+ table does not exist or is not accessible, including when the catalog does not exist.
157
+ Every other failure makes the affected facts unavailable; for a view, the history facts
158
+ are unavailable.
159
+ """
160
+ ref = parse_table(table)
161
+ validate_limit(limit)
162
+ runner = get_runner(spark)
163
+
164
+ info = collect_table_info(runner, ref)
165
+ time_zone = collect_session_time_zone(runner)
166
+ result = collect_commits(runner, ref, info.object_type, limit)
167
+ return build_history_analysis(ref, limit, time_zone, result.commits, result.metrics_reason)
@@ -0,0 +1,80 @@
1
+ """Collection layer: queries to raw data to model objects. The only layer that knows SQL."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from typing import Any
7
+
8
+ from dataowl.model.facts import Fact
9
+ from dataowl.model.overview import ObjectType
10
+ from dataowl.runner import SqlRunner
11
+
12
+ _MAX_REASON_LENGTH = 200
13
+
14
+ NOT_FOR_VIEWS = "Not available for views"
15
+
16
+ _CATALOG_NOT_FOUND_CLASSES = frozenset({"NO_SUCH_CATALOG_EXCEPTION", "CATALOG_NOT_FOUND"})
17
+ _TABLE_OR_VIEW_NOT_FOUND = "TABLE_OR_VIEW_NOT_FOUND"
18
+ _LEADING_ERROR_CLASS = re.compile(r"\[([A-Za-z0-9_.]+)\]")
19
+ _INFORMATION_SCHEMA = re.compile(r"(?<!\w)information_schema(?!\w)", re.IGNORECASE)
20
+
21
+
22
+ def is_view(object_type: Fact[ObjectType]) -> bool:
23
+ """True only when the object type is known to be VIEW."""
24
+ return object_type.available and object_type.value is ObjectType.VIEW
25
+
26
+
27
+ def safe_query(
28
+ runner: SqlRunner, sql: str, params: dict[str, Any] | None = None
29
+ ) -> list[dict[str, Any]] | Exception:
30
+ """Run a query and return the rows, or the exception instead of raising it."""
31
+ try:
32
+ return runner.query(sql, params)
33
+ except Exception as exc:
34
+ return exc
35
+
36
+
37
+ def is_catalog_not_found(exc: Exception) -> bool:
38
+ """True when the error says that the catalog does not exist.
39
+
40
+ The error class is read from getCondition(), then getErrorClass(), and otherwise from a
41
+ leading "[CLASS]" in the first line of the message. PySpark is not imported. True for
42
+ NO_SUCH_CATALOG_EXCEPTION and CATALOG_NOT_FOUND, and for TABLE_OR_VIEW_NOT_FOUND when the
43
+ first line names information_schema, quoted or not. False for everything else, including
44
+ access errors.
45
+ """
46
+ lines = str(exc).strip().splitlines()
47
+ first_line = lines[0].strip() if lines else ""
48
+ error_class = _error_class(exc, first_line)
49
+ if error_class in _CATALOG_NOT_FOUND_CLASSES:
50
+ return True
51
+ if error_class == _TABLE_OR_VIEW_NOT_FOUND:
52
+ return _INFORMATION_SCHEMA.search(first_line) is not None
53
+ return False
54
+
55
+
56
+ def _error_class(exc: Exception, first_line: str) -> str | None:
57
+ for method_name in ("getCondition", "getErrorClass"):
58
+ method = getattr(exc, method_name, None)
59
+ if not callable(method):
60
+ continue
61
+ try:
62
+ value = method()
63
+ except Exception:
64
+ continue
65
+ if isinstance(value, str) and value.strip():
66
+ return value.strip().upper()
67
+ match = _LEADING_ERROR_CLASS.match(first_line)
68
+ return match.group(1).upper() if match else None
69
+
70
+
71
+ def short_reason(exc: Exception) -> str:
72
+ """First line of the error message, at most 200 characters.
73
+
74
+ Falls back to the exception type name when the message is empty.
75
+ """
76
+ lines = str(exc).strip().splitlines()
77
+ reason = lines[0].strip() if lines else type(exc).__name__
78
+ if len(reason) > _MAX_REASON_LENGTH:
79
+ reason = reason[: _MAX_REASON_LENGTH - 1] + "…"
80
+ return reason