dataowl 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dataowl/__init__.py ADDED
@@ -0,0 +1,62 @@
1
+ """dataowl: facts about Databricks data products for building dbt staging models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING
6
+
7
+ from dataowl.collect.columns import collect_columns
8
+ from dataowl.collect.counts import collect_row_count
9
+ from dataowl.collect.detail import collect_detail
10
+ from dataowl.collect.properties import collect_properties
11
+ from dataowl.collect.tables import collect_table_info
12
+ from dataowl.errors import TableNotFoundError
13
+ from dataowl.identifiers import parse_table
14
+ from dataowl.model.overview import Overview
15
+ from dataowl.runner import get_runner
16
+
17
+ if TYPE_CHECKING:
18
+ from pyspark.sql import SparkSession
19
+
20
+ __all__ = ["Overview", "TableNotFoundError", "inspect"]
21
+
22
+
23
+ def inspect(
24
+ table: str, *, spark: SparkSession | None = None, count_views: bool = False
25
+ ) -> Overview:
26
+ """Collect overview facts for a table.
27
+
28
+ Raises ValueError for an invalid table name, RuntimeError when no SparkSession is
29
+ available, and TableNotFoundError when the table does not exist or is not accessible.
30
+ Every other failure makes the affected facts unavailable.
31
+ """
32
+ ref = parse_table(table)
33
+ runner = get_runner(spark)
34
+
35
+ info = collect_table_info(runner, ref)
36
+ detail = collect_detail(runner, ref, info.object_type)
37
+ columns = collect_columns(runner, ref)
38
+ properties = collect_properties(runner, ref, info.object_type)
39
+ num_rows = collect_row_count(runner, ref, info.object_type, count_views=count_views)
40
+
41
+ return Overview(
42
+ table=ref,
43
+ object_type=info.object_type,
44
+ object_type_raw=info.table_type_raw,
45
+ format=info.format,
46
+ owner=info.owner,
47
+ comment=info.comment,
48
+ created=info.created,
49
+ last_modified=detail.last_modified,
50
+ size_bytes=detail.size_bytes,
51
+ num_files=detail.num_files,
52
+ avg_file_size_bytes=detail.avg_file_size_bytes,
53
+ num_rows=num_rows,
54
+ num_columns=columns.num_columns,
55
+ num_fields_nested=columns.num_fields_nested,
56
+ partition_columns=detail.partition_columns,
57
+ clustering_columns=detail.clustering_columns,
58
+ change_data_feed=properties.change_data_feed,
59
+ log_retention=properties.log_retention,
60
+ deleted_file_retention=properties.deleted_file_retention,
61
+ columns=columns.columns,
62
+ )
@@ -0,0 +1,40 @@
1
+ """Collection layer: queries to raw data to model objects. The only layer that knows SQL."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from dataowl.model.facts import Fact
8
+ from dataowl.model.overview import ObjectType
9
+ from dataowl.runner import SqlRunner
10
+
11
+ _MAX_REASON_LENGTH = 200
12
+
13
+ NOT_FOR_VIEWS = "Not available for views"
14
+
15
+
16
+ def is_view(object_type: Fact[ObjectType]) -> bool:
17
+ """True only when the object type is known to be VIEW."""
18
+ return object_type.available and object_type.value is ObjectType.VIEW
19
+
20
+
21
+ def safe_query(
22
+ runner: SqlRunner, sql: str, params: dict[str, Any] | None = None
23
+ ) -> list[dict[str, Any]] | Exception:
24
+ """Run a query and return the rows, or the exception instead of raising it."""
25
+ try:
26
+ return runner.query(sql, params)
27
+ except Exception as exc:
28
+ return exc
29
+
30
+
31
+ def short_reason(exc: Exception) -> str:
32
+ """First line of the error message, at most 200 characters.
33
+
34
+ Falls back to the exception type name when the message is empty.
35
+ """
36
+ lines = str(exc).strip().splitlines()
37
+ reason = lines[0].strip() if lines else type(exc).__name__
38
+ if len(reason) > _MAX_REASON_LENGTH:
39
+ reason = reason[: _MAX_REASON_LENGTH - 1] + "…"
40
+ return reason
@@ -0,0 +1,135 @@
1
+ """Schema facts from information_schema.columns and the Spark schema."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING, Any, TypeVar
6
+
7
+ from dataowl.collect import safe_query, short_reason
8
+ from dataowl.identifiers import TableRef
9
+ from dataowl.model.facts import Fact
10
+ from dataowl.model.overview import ColumnInfo, ColumnsInfo
11
+ from dataowl.runner import SqlRunner
12
+
13
+ if TYPE_CHECKING:
14
+ from pyspark.sql.types import DataType, StructType
15
+
16
+ T = TypeVar("T")
17
+
18
+ NO_COLUMNS = "information_schema.columns returned no rows"
19
+
20
+ _QUERY = """
21
+ SELECT
22
+ column_name AS column_name,
23
+ ordinal_position AS ordinal_position,
24
+ full_data_type AS full_data_type,
25
+ data_type AS data_type,
26
+ is_nullable AS is_nullable,
27
+ comment AS comment
28
+ FROM {catalog}.information_schema.columns
29
+ WHERE table_schema = :schema
30
+ AND table_name = :table
31
+ ORDER BY ordinal_position
32
+ """
33
+
34
+
35
+ def collect_columns(runner: SqlRunner, ref: TableRef) -> ColumnsInfo:
36
+ """Read top-level columns from information_schema and count nested fields from the schema.
37
+
38
+ The two sources fail independently.
39
+ """
40
+ columns, num_columns = _top_level_columns(runner, ref)
41
+ return ColumnsInfo(
42
+ columns=columns,
43
+ num_columns=num_columns,
44
+ num_fields_nested=_nested_field_count(runner, ref),
45
+ )
46
+
47
+
48
+ def count_fields(schema: StructType) -> int:
49
+ """Count all fields in a schema, including nested fields.
50
+
51
+ Every StructField counts 1. Fields inside a struct, inside the element type of an
52
+ array and inside the key and value types of a map are added recursively.
53
+
54
+ Example: ``a INT, b STRUCT<x INT, y ARRAY<STRUCT<z INT>>>, m MAP<STRING, STRUCT<v INT>>``
55
+ counts a, b, x, y, z, m and v, which gives 7.
56
+ """
57
+ from pyspark.sql.types import ArrayType, MapType, StructType
58
+
59
+ def nested(data_type: DataType) -> int:
60
+ if isinstance(data_type, StructType):
61
+ return sum(1 + nested(field.dataType) for field in data_type.fields)
62
+ if isinstance(data_type, ArrayType):
63
+ return nested(data_type.elementType)
64
+ if isinstance(data_type, MapType):
65
+ return nested(data_type.keyType) + nested(data_type.valueType)
66
+ return 0
67
+
68
+ return nested(schema)
69
+
70
+
71
+ def _top_level_columns(
72
+ runner: SqlRunner, ref: TableRef
73
+ ) -> tuple[Fact[tuple[ColumnInfo, ...]], Fact[int]]:
74
+ result = safe_query(
75
+ runner,
76
+ _QUERY.format(catalog=ref.quoted_catalog()),
77
+ # Unity Catalog stores names in lower case. Lowering the parameters instead of
78
+ # the columns keeps the WHERE clause free of functions on information_schema columns.
79
+ {"schema": ref.schema.lower(), "table": ref.table.lower()},
80
+ )
81
+ if isinstance(result, Exception):
82
+ reason = short_reason(result)
83
+ return Fact.unavailable("metadata", reason), Fact.unavailable("metadata", reason)
84
+ if not result:
85
+ return Fact.unavailable("metadata", NO_COLUMNS), Fact.unavailable("metadata", NO_COLUMNS)
86
+
87
+ try:
88
+ columns = tuple(_column(row) for row in result)
89
+ except ValueError as exc:
90
+ reason = f"Unexpected row format in information_schema.columns: {exc}"
91
+ return Fact.unavailable("metadata", reason), Fact.unavailable("metadata", reason)
92
+ return Fact(columns, source="metadata"), Fact(len(columns), source="metadata")
93
+
94
+
95
+ def _column(row: dict[str, Any]) -> ColumnInfo:
96
+ """Convert one row. Raises ValueError if a key is missing or has an unexpected type."""
97
+ name = _get(row, "column_name", str)
98
+ position = _get(row, "ordinal_position", int)
99
+ full_data_type = row.get("full_data_type")
100
+ if full_data_type is None:
101
+ data_type = _get(row, "data_type", str)
102
+ elif isinstance(full_data_type, str):
103
+ data_type = full_data_type
104
+ else:
105
+ raise ValueError(f"full_data_type has type {type(full_data_type).__name__}")
106
+ is_nullable = _get(row, "is_nullable", str)
107
+ if is_nullable not in ("YES", "NO"):
108
+ raise ValueError(f"is_nullable has value {is_nullable!r}")
109
+ comment = row.get("comment")
110
+ if comment is not None and not isinstance(comment, str):
111
+ raise ValueError(f"comment has type {type(comment).__name__}")
112
+ return ColumnInfo(
113
+ name=name,
114
+ position=position,
115
+ data_type=data_type,
116
+ nullable=is_nullable == "YES",
117
+ comment=comment,
118
+ )
119
+
120
+
121
+ def _get(row: dict[str, Any], key: str, expected: type[T]) -> T:
122
+ if key not in row:
123
+ raise ValueError(f"missing {key}")
124
+ value = row[key]
125
+ if not isinstance(value, expected) or isinstance(value, bool):
126
+ raise ValueError(f"{key} has type {type(value).__name__}")
127
+ return value
128
+
129
+
130
+ def _nested_field_count(runner: SqlRunner, ref: TableRef) -> Fact[int]:
131
+ try:
132
+ schema = runner.schema(ref)
133
+ except Exception as exc:
134
+ return Fact.unavailable("metadata", short_reason(exc))
135
+ return Fact(count_fields(schema), source="metadata")
@@ -0,0 +1,46 @@
1
+ """Exact row count."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataowl.collect import safe_query, short_reason
6
+ from dataowl.identifiers import TableRef
7
+ from dataowl.model.facts import Fact
8
+ from dataowl.model.overview import ObjectType
9
+ from dataowl.runner import SqlRunner
10
+
11
+ _SKIPPED_TYPES = frozenset({ObjectType.VIEW, ObjectType.MATERIALIZED_VIEW, ObjectType.FOREIGN})
12
+
13
+ NO_ROWS = "COUNT(*) returned no rows"
14
+
15
+
16
+ def collect_row_count(
17
+ runner: SqlRunner,
18
+ ref: TableRef,
19
+ object_type: Fact[ObjectType],
20
+ *,
21
+ count_views: bool = False,
22
+ ) -> Fact[int]:
23
+ """Count rows with COUNT(*).
24
+
25
+ Unless count_views is True, the count is skipped for views, materialized views and
26
+ foreign tables, and when the object type is unavailable. UNKNOWN is counted.
27
+ """
28
+ if not count_views:
29
+ if not object_type.available:
30
+ return Fact.unavailable("exact", "Skipped: object type unknown; use count_views=True")
31
+ if object_type.value in _SKIPPED_TYPES:
32
+ assert object_type.value is not None
33
+ return Fact.unavailable(
34
+ "exact", f"Skipped for {object_type.value.value}; use count_views=True"
35
+ )
36
+
37
+ result = safe_query(runner, f"SELECT COUNT(*) AS n FROM {ref.quoted()}")
38
+ if isinstance(result, Exception):
39
+ return Fact.unavailable("exact", short_reason(result))
40
+ if not result:
41
+ return Fact.unavailable("exact", NO_ROWS)
42
+
43
+ n = result[0].get("n")
44
+ if not isinstance(n, int) or isinstance(n, bool):
45
+ return Fact.unavailable("exact", f"Unexpected COUNT(*) result type: {type(n).__name__}")
46
+ return Fact(n, source="exact")
@@ -0,0 +1,80 @@
1
+ """Size, files and Delta metadata from DESCRIBE DETAIL."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from dataowl.collect import NOT_FOR_VIEWS, is_view, safe_query, short_reason
8
+ from dataowl.identifiers import TableRef
9
+ from dataowl.model.facts import Fact, derive
10
+ from dataowl.model.overview import DetailInfo, ObjectType
11
+ from dataowl.runner import SqlRunner
12
+
13
+ NOT_PRESENT = "Not present in DESCRIBE DETAIL output"
14
+ NO_ROWS = "DESCRIBE DETAIL returned no rows"
15
+
16
+
17
+ def collect_detail(runner: SqlRunner, ref: TableRef, object_type: Fact[ObjectType]) -> DetailInfo:
18
+ """Run DESCRIBE DETAIL, except for views.
19
+
20
+ It also runs when the object type is UNKNOWN or unavailable. Errors make every fact
21
+ unavailable.
22
+ """
23
+ if is_view(object_type):
24
+ return _unavailable(NOT_FOR_VIEWS)
25
+
26
+ result = safe_query(runner, f"DESCRIBE DETAIL {ref.quoted()}")
27
+ if isinstance(result, Exception):
28
+ return _unavailable(short_reason(result))
29
+ if not result:
30
+ return _unavailable(NO_ROWS)
31
+
32
+ row = result[0]
33
+ size_bytes: Fact[int] = _field(row, "sizeInBytes")
34
+ num_files: Fact[int] = _field(row, "numFiles")
35
+ return DetailInfo(
36
+ format=_field(row, "format"),
37
+ size_bytes=size_bytes,
38
+ num_files=num_files,
39
+ avg_file_size_bytes=_avg_file_size(size_bytes, num_files),
40
+ created=_field(row, "createdAt"),
41
+ last_modified=_field(row, "lastModified"),
42
+ partition_columns=_columns(row, "partitionColumns"),
43
+ clustering_columns=_columns(row, "clusteringColumns"),
44
+ )
45
+
46
+
47
+ def _field(row: dict[str, Any], key: str) -> Fact[Any]:
48
+ if key not in row:
49
+ return Fact.unavailable("metadata", NOT_PRESENT)
50
+ return Fact(row[key], source="metadata")
51
+
52
+
53
+ def _columns(row: dict[str, Any], key: str) -> Fact[tuple[str, ...]]:
54
+ fact: Fact[Any] = _field(row, key)
55
+ if fact.available and fact.value is not None:
56
+ return Fact(tuple(fact.value), source="metadata")
57
+ return fact
58
+
59
+
60
+ def _avg_file_size(size_bytes: Fact[int], num_files: Fact[int]) -> Fact[float]:
61
+ if num_files.available and num_files.value == 0:
62
+ return Fact.unavailable("derived", "No files")
63
+ if (size_bytes.available and size_bytes.value is None) or (
64
+ num_files.available and num_files.value is None
65
+ ):
66
+ return Fact.unavailable("derived", "Input value is null")
67
+ return derive(lambda size, files: size / files, size_bytes, num_files)
68
+
69
+
70
+ def _unavailable(reason: str) -> DetailInfo:
71
+ return DetailInfo(
72
+ format=Fact.unavailable("metadata", reason),
73
+ size_bytes=Fact.unavailable("metadata", reason),
74
+ num_files=Fact.unavailable("metadata", reason),
75
+ avg_file_size_bytes=Fact.unavailable("derived", reason),
76
+ created=Fact.unavailable("metadata", reason),
77
+ last_modified=Fact.unavailable("metadata", reason),
78
+ partition_columns=Fact.unavailable("metadata", reason),
79
+ clustering_columns=Fact.unavailable("metadata", reason),
80
+ )
@@ -0,0 +1 @@
1
+ from __future__ import annotations
@@ -0,0 +1 @@
1
+ from __future__ import annotations
@@ -0,0 +1,75 @@
1
+ """Selected Delta table properties from SHOW TBLPROPERTIES."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from dataowl.collect import NOT_FOR_VIEWS, is_view, safe_query, short_reason
8
+ from dataowl.identifiers import TableRef
9
+ from dataowl.model.facts import Fact
10
+ from dataowl.model.overview import ObjectType, PropertiesInfo
11
+ from dataowl.runner import SqlRunner
12
+
13
+ _CHANGE_DATA_FEED = "delta.enablechangedatafeed"
14
+ _LOG_RETENTION = "delta.logretentionduration"
15
+ _DELETED_FILE_RETENTION = "delta.deletedfileretentionduration"
16
+ _KEYS = frozenset({_CHANGE_DATA_FEED, _LOG_RETENTION, _DELETED_FILE_RETENTION})
17
+
18
+
19
+ def collect_properties(
20
+ runner: SqlRunner, ref: TableRef, object_type: Fact[ObjectType]
21
+ ) -> PropertiesInfo:
22
+ """Run SHOW TBLPROPERTIES, except for views.
23
+
24
+ It also runs when the object type is UNKNOWN or unavailable. Keys are matched case
25
+ insensitively. A property that is not set gives Fact(None, source="metadata").
26
+ """
27
+ if is_view(object_type):
28
+ return _unavailable(NOT_FOR_VIEWS)
29
+
30
+ result = safe_query(runner, f"SHOW TBLPROPERTIES {ref.quoted()}")
31
+ if isinstance(result, Exception):
32
+ return _unavailable(short_reason(result))
33
+
34
+ facts = _property_facts(result)
35
+ return PropertiesInfo(
36
+ change_data_feed=facts[_CHANGE_DATA_FEED],
37
+ log_retention=facts[_LOG_RETENTION],
38
+ deleted_file_retention=facts[_DELETED_FILE_RETENTION],
39
+ )
40
+
41
+
42
+ def _property_facts(rows: list[dict[str, Any]]) -> dict[str, Fact[str]]:
43
+ """Return one fact per relevant property, keyed by lower-case key.
44
+
45
+ Rows whose key is not a string, and rows for other properties, are skipped. A relevant
46
+ row whose value is not a string makes only that property unavailable.
47
+ """
48
+ facts: dict[str, Fact[str]] = {key: Fact(None, source="metadata") for key in _KEYS}
49
+ seen: set[str] = set()
50
+ for row in rows:
51
+ key = row.get("key")
52
+ if not isinstance(key, str):
53
+ continue
54
+ normalized = key.lower()
55
+ if normalized not in _KEYS or normalized in seen:
56
+ continue
57
+ seen.add(normalized)
58
+ value = row.get("value")
59
+ if isinstance(value, str):
60
+ facts[normalized] = Fact(value, source="metadata")
61
+ else:
62
+ facts[normalized] = Fact.unavailable(
63
+ "metadata",
64
+ f"Unexpected row format in SHOW TBLPROPERTIES: value of {key} has type "
65
+ f"{type(value).__name__}",
66
+ )
67
+ return facts
68
+
69
+
70
+ def _unavailable(reason: str) -> PropertiesInfo:
71
+ return PropertiesInfo(
72
+ change_data_feed=Fact.unavailable("metadata", reason),
73
+ log_retention=Fact.unavailable("metadata", reason),
74
+ deleted_file_retention=Fact.unavailable("metadata", reason),
75
+ )
@@ -0,0 +1,80 @@
1
+ """Object type and basic catalog metadata from information_schema.tables."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from dataowl.collect import safe_query, short_reason
8
+ from dataowl.errors import TableNotFoundError
9
+ from dataowl.identifiers import TableRef
10
+ from dataowl.model.facts import Fact
11
+ from dataowl.model.overview import ObjectType, TableInfo
12
+ from dataowl.runner import SqlRunner
13
+
14
+ _QUERY = """
15
+ SELECT
16
+ table_type AS table_type,
17
+ data_source_format AS data_source_format,
18
+ table_owner AS table_owner,
19
+ comment AS comment,
20
+ created AS created,
21
+ last_altered AS last_altered
22
+ FROM {catalog}.information_schema.tables
23
+ WHERE table_schema = :schema
24
+ AND table_name = :table
25
+ """
26
+
27
+
28
+ def collect_table_info(runner: SqlRunner, ref: TableRef) -> TableInfo:
29
+ """Read catalog metadata for the table.
30
+
31
+ Raises TableNotFoundError if the query succeeds but returns no rows. Any other
32
+ failure makes every fact unavailable.
33
+ """
34
+ result = safe_query(
35
+ runner,
36
+ _QUERY.format(catalog=ref.quoted_catalog()),
37
+ # Unity Catalog stores names in lower case. Lowering the parameters instead of
38
+ # the columns keeps the WHERE clause free of functions on information_schema columns.
39
+ {"schema": ref.schema.lower(), "table": ref.table.lower()},
40
+ )
41
+
42
+ if isinstance(result, Exception):
43
+ reason = short_reason(result)
44
+ return TableInfo(
45
+ object_type=Fact.unavailable("metadata", reason),
46
+ table_type_raw=Fact.unavailable("metadata", reason),
47
+ format=Fact.unavailable("metadata", reason),
48
+ owner=Fact.unavailable("metadata", reason),
49
+ comment=Fact.unavailable("metadata", reason),
50
+ created=Fact.unavailable("metadata", reason),
51
+ last_altered=Fact.unavailable("metadata", reason),
52
+ )
53
+
54
+ if not result:
55
+ raise TableNotFoundError(f"Table {ref.quoted()} not found or no access")
56
+
57
+ row = result[0]
58
+ raw_type = row.get("table_type")
59
+ return TableInfo(
60
+ object_type=Fact(_object_type(raw_type), source="metadata"),
61
+ table_type_raw=_metadata(raw_type),
62
+ format=_metadata(row.get("data_source_format")),
63
+ owner=_metadata(row.get("table_owner")),
64
+ comment=_metadata(row.get("comment")),
65
+ created=_metadata(row.get("created")),
66
+ last_altered=_metadata(row.get("last_altered")),
67
+ )
68
+
69
+
70
+ def _object_type(raw: str | None) -> ObjectType:
71
+ if raw is None:
72
+ return ObjectType.UNKNOWN
73
+ try:
74
+ return ObjectType(raw.strip().upper())
75
+ except ValueError:
76
+ return ObjectType.UNKNOWN
77
+
78
+
79
+ def _metadata(value: Any) -> Fact[Any]:
80
+ return Fact(value, source="metadata")
@@ -0,0 +1 @@
1
+ from __future__ import annotations
dataowl/errors.py ADDED
@@ -0,0 +1,7 @@
1
+ """Exceptions raised by dataowl."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class TableNotFoundError(Exception):
7
+ """The table does not exist, or the current user has no access to it."""
dataowl/identifiers.py ADDED
@@ -0,0 +1,112 @@
1
+ """Parsing, validation and quoting of table and column names."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import string
6
+ from dataclasses import dataclass
7
+
8
+ _UNQUOTED_CHARS = frozenset(string.ascii_letters + string.digits + "_")
9
+
10
+
11
+ @dataclass(frozen=True)
12
+ class TableRef:
13
+ """A three-part table name. Parts are stored raw (unquoted), with case preserved."""
14
+
15
+ catalog: str
16
+ schema: str
17
+ table: str
18
+
19
+ def __post_init__(self) -> None:
20
+ if not (self.catalog and self.schema and self.table):
21
+ raise ValueError("TableRef parts must be non-empty")
22
+
23
+ def quoted(self) -> str:
24
+ return ".".join(_quote(part) for part in (self.catalog, self.schema, self.table))
25
+
26
+ def quoted_catalog(self) -> str:
27
+ return _quote(self.catalog)
28
+
29
+ def display_name(self) -> str:
30
+ """Name for display: parts with only letters, digits and underscore are unquoted."""
31
+ return ".".join(
32
+ part if all(char in _UNQUOTED_CHARS for char in part) else _quote(part)
33
+ for part in (self.catalog, self.schema, self.table)
34
+ )
35
+
36
+
37
+ def parse_table(name: str) -> TableRef:
38
+ """Parse `catalog.schema.table`. Parts may be quoted with backticks."""
39
+ if not name:
40
+ raise ValueError("Table name is empty")
41
+ if name != name.strip():
42
+ raise ValueError(f"Table name has leading or trailing whitespace: {name!r}")
43
+
44
+ parts: list[str] = []
45
+ i = 0
46
+ while True:
47
+ if i < len(name) and name[i] == "`":
48
+ part, i = _read_quoted(name, i)
49
+ else:
50
+ part, i = _read_unquoted(name, i)
51
+ if not part:
52
+ raise ValueError(f"Table name has an empty part: {name!r}")
53
+ parts.append(part)
54
+
55
+ if i == len(name):
56
+ break
57
+ if name[i] != ".":
58
+ raise ValueError(
59
+ f"Unexpected character {name[i]!r} after closing backtick in table name: {name!r}"
60
+ )
61
+ i += 1
62
+
63
+ if len(parts) != 3:
64
+ raise ValueError(
65
+ f"Table name must have exactly three parts (catalog.schema.table), "
66
+ f"got {len(parts)}: {name!r}"
67
+ )
68
+ return TableRef(*parts)
69
+
70
+
71
+ def quote_column(name: str) -> str:
72
+ """Quote a column name as a single identifier. Dots are not treated as separators."""
73
+ if not name:
74
+ raise ValueError("Column name is empty")
75
+ return _quote(name)
76
+
77
+
78
+ def _quote(identifier: str) -> str:
79
+ return "`" + identifier.replace("`", "``") + "`"
80
+
81
+
82
+ def _read_quoted(name: str, start: int) -> tuple[str, int]:
83
+ """Read a backtick-quoted part starting at the opening backtick.
84
+
85
+ Returns the raw part and the index after the closing backtick.
86
+ """
87
+ chars: list[str] = []
88
+ i = start + 1
89
+ while i < len(name):
90
+ if name[i] == "`":
91
+ if i + 1 < len(name) and name[i + 1] == "`":
92
+ chars.append("`")
93
+ i += 2
94
+ continue
95
+ return "".join(chars), i + 1
96
+ chars.append(name[i])
97
+ i += 1
98
+ raise ValueError(f"Unterminated backtick in table name: {name!r}")
99
+
100
+
101
+ def _read_unquoted(name: str, start: int) -> tuple[str, int]:
102
+ """Read an unquoted part. Returns the part and the index of the next '.' or the end."""
103
+ i = start
104
+ while i < len(name) and name[i] != ".":
105
+ if name[i] not in _UNQUOTED_CHARS:
106
+ raise ValueError(
107
+ f"Invalid character {name[i]!r} in unquoted part of table name: {name!r}. "
108
+ "Use backticks for names with characters other than letters, digits "
109
+ "and underscore."
110
+ )
111
+ i += 1
112
+ return name[start:i], i
@@ -0,0 +1,3 @@
1
+ """Data model. Dataclasses only; no Spark import."""
2
+
3
+ from __future__ import annotations
@@ -0,0 +1 @@
1
+ from __future__ import annotations
dataowl/model/facts.py ADDED
@@ -0,0 +1,75 @@
1
+ """Fact: a single value together with its source and availability."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from dataclasses import dataclass, fields, is_dataclass
7
+ from datetime import date
8
+ from enum import Enum
9
+ from typing import Any, Generic, Literal, TypeVar
10
+
11
+ T = TypeVar("T")
12
+ R = TypeVar("R")
13
+
14
+ SourceType = Literal["metadata", "exact", "derived"]
15
+
16
+
17
+ @dataclass(frozen=True)
18
+ class Fact(Generic[T]):
19
+ """A fact with its source.
20
+
21
+ An unavailable fact requires value None and a non-empty reason. An available fact
22
+ requires reason None. An available fact may still have value None, e.g. a table
23
+ property that is not set.
24
+ """
25
+
26
+ value: T | None
27
+ source: SourceType
28
+ available: bool = True
29
+ reason: str | None = None
30
+
31
+ def __post_init__(self) -> None:
32
+ if self.available and self.reason is not None:
33
+ raise ValueError("An available fact must not have a reason")
34
+ if not self.available and not self.reason:
35
+ raise ValueError("An unavailable fact requires a non-empty reason")
36
+ if not self.available and self.value is not None:
37
+ raise ValueError("An unavailable fact must have value None")
38
+
39
+ @classmethod
40
+ def unavailable(cls, source: SourceType, reason: str) -> Fact[T]:
41
+ return cls(value=None, source=source, available=False, reason=reason)
42
+
43
+
44
+ def derive(fn: Callable[..., R], *facts: Fact[Any]) -> Fact[R]:
45
+ """Apply fn to the values of facts, or propagate the first unavailable input."""
46
+ for fact in facts:
47
+ if not fact.available:
48
+ assert fact.reason is not None # guaranteed by Fact.__post_init__
49
+ return Fact.unavailable("derived", fact.reason)
50
+ return Fact(fn(*(fact.value for fact in facts)), source="derived")
51
+
52
+
53
+ def to_jsonable(obj: Any) -> Any:
54
+ """Convert facts and model objects to JSON-serializable types.
55
+
56
+ Fact becomes {"value", "source", "available", "reason"}. Other dataclasses become a
57
+ dict of their fields. Tuples and lists become lists, enums their value, and datetime
58
+ and date ISO 8601 strings. Other values are returned unchanged.
59
+ """
60
+ if isinstance(obj, Fact):
61
+ return {
62
+ "value": to_jsonable(obj.value),
63
+ "source": obj.source,
64
+ "available": obj.available,
65
+ "reason": obj.reason,
66
+ }
67
+ if is_dataclass(obj) and not isinstance(obj, type):
68
+ return {field.name: to_jsonable(getattr(obj, field.name)) for field in fields(obj)}
69
+ if isinstance(obj, (tuple, list)):
70
+ return [to_jsonable(item) for item in obj]
71
+ if isinstance(obj, Enum):
72
+ return obj.value
73
+ if isinstance(obj, date):
74
+ return obj.isoformat()
75
+ return obj
@@ -0,0 +1 @@
1
+ from __future__ import annotations
@@ -0,0 +1,130 @@
1
+ """Overview model: facts from step 1 (inspect)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from datetime import datetime
7
+ from enum import Enum
8
+ from typing import Any
9
+
10
+ from dataowl.identifiers import TableRef
11
+ from dataowl.model.facts import Fact, to_jsonable
12
+
13
+
14
+ class ObjectType(Enum):
15
+ MANAGED = "MANAGED"
16
+ EXTERNAL = "EXTERNAL"
17
+ VIEW = "VIEW"
18
+ MATERIALIZED_VIEW = "MATERIALIZED_VIEW"
19
+ STREAMING_TABLE = "STREAMING_TABLE"
20
+ FOREIGN = "FOREIGN"
21
+ UNKNOWN = "UNKNOWN"
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class TableInfo:
26
+ """Catalog metadata from information_schema.tables.
27
+
28
+ table_type_raw holds the original table_type value, also when object_type is UNKNOWN.
29
+ """
30
+
31
+ object_type: Fact[ObjectType]
32
+ table_type_raw: Fact[str]
33
+ format: Fact[str]
34
+ owner: Fact[str]
35
+ comment: Fact[str]
36
+ created: Fact[datetime]
37
+ last_altered: Fact[datetime]
38
+
39
+
40
+ @dataclass(frozen=True)
41
+ class DetailInfo:
42
+ """Size, files and Delta metadata from DESCRIBE DETAIL."""
43
+
44
+ format: Fact[str]
45
+ size_bytes: Fact[int]
46
+ num_files: Fact[int]
47
+ avg_file_size_bytes: Fact[float]
48
+ created: Fact[datetime]
49
+ last_modified: Fact[datetime]
50
+ partition_columns: Fact[tuple[str, ...]]
51
+ clustering_columns: Fact[tuple[str, ...]]
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class PropertiesInfo:
56
+ """Selected Delta table properties from SHOW TBLPROPERTIES, as raw strings.
57
+
58
+ A property that is not set is an available fact with value None, e.g.
59
+ Fact(None, source="metadata"). The model holds no display text for this; rendering
60
+ decides how to show it.
61
+ """
62
+
63
+ change_data_feed: Fact[str]
64
+ log_retention: Fact[str]
65
+ deleted_file_retention: Fact[str]
66
+
67
+
68
+ @dataclass(frozen=True)
69
+ class ColumnInfo:
70
+ """One top-level column from information_schema.columns.
71
+
72
+ position is the raw ordinal_position from information_schema.columns. The Databricks
73
+ documentation says it is numbered from 1, but it has been observed to be 0-based in
74
+ Databricks (serverless, October 2026). Do not use it as a display number.
75
+ """
76
+
77
+ name: str
78
+ position: int
79
+ data_type: str
80
+ nullable: bool
81
+ comment: str | None
82
+
83
+
84
+ @dataclass(frozen=True)
85
+ class ColumnsInfo:
86
+ """Schema facts: top-level columns and the field count including nested fields."""
87
+
88
+ columns: Fact[tuple[ColumnInfo, ...]]
89
+ num_columns: Fact[int]
90
+ num_fields_nested: Fact[int]
91
+
92
+
93
+ @dataclass(frozen=True)
94
+ class Overview:
95
+ """All facts from inspect().
96
+
97
+ format and created come from information_schema; last_modified is the last data change
98
+ from DESCRIBE DETAIL. object_type_raw is the original table_type, also when object_type
99
+ is UNKNOWN.
100
+ """
101
+
102
+ table: TableRef
103
+ object_type: Fact[ObjectType]
104
+ object_type_raw: Fact[str]
105
+ format: Fact[str]
106
+ owner: Fact[str]
107
+ comment: Fact[str]
108
+ created: Fact[datetime]
109
+ last_modified: Fact[datetime]
110
+ size_bytes: Fact[int]
111
+ num_files: Fact[int]
112
+ avg_file_size_bytes: Fact[float]
113
+ num_rows: Fact[int]
114
+ num_columns: Fact[int]
115
+ num_fields_nested: Fact[int]
116
+ partition_columns: Fact[tuple[str, ...]]
117
+ clustering_columns: Fact[tuple[str, ...]]
118
+ change_data_feed: Fact[str]
119
+ log_retention: Fact[str]
120
+ deleted_file_retention: Fact[str]
121
+ columns: Fact[tuple[ColumnInfo, ...]]
122
+
123
+ def to_dict(self) -> dict[str, Any]:
124
+ result: dict[str, Any] = to_jsonable(self)
125
+ return result
126
+
127
+ def show(self) -> None:
128
+ from dataowl.render.terminal import render_overview
129
+
130
+ print(render_overview(self))
dataowl/py.typed ADDED
File without changes
@@ -0,0 +1,3 @@
1
+ """Rendering to text. No Spark import."""
2
+
3
+ from __future__ import annotations
@@ -0,0 +1,44 @@
1
+ """Formatting of single values for display."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from datetime import datetime
7
+ from typing import TypeVar
8
+
9
+ from dataowl.model.facts import Fact
10
+
11
+ T = TypeVar("T")
12
+
13
+ _UNITS = ("KB", "MB", "GB", "TB")
14
+
15
+
16
+ def format_bytes(n: float) -> str:
17
+ """Bytes below 1024 as an integer with B, then KB to TB with one decimal (base 1024)."""
18
+ if round(n) < 1024:
19
+ return f"{round(n)} B"
20
+ value = float(n)
21
+ for unit in _UNITS:
22
+ value /= 1024
23
+ if round(value, 1) < 1024 or unit == _UNITS[-1]:
24
+ return f"{value:.1f} {unit}"
25
+ raise AssertionError("unreachable")
26
+
27
+
28
+ def format_int(n: int) -> str:
29
+ """Integer with a space as thousands separator."""
30
+ return f"{n:,}".replace(",", " ")
31
+
32
+
33
+ def format_datetime(value: datetime) -> str:
34
+ """YYYY-MM-DD HH:MM, without time zone conversion."""
35
+ return value.strftime("%Y-%m-%d %H:%M")
36
+
37
+
38
+ def format_fact(fact: Fact[T], fmt: Callable[[T], str] | None = None, none_text: str = "–") -> str:
39
+ """Format a fact: 'n/a (<reason>)' if unavailable, none_text if the value is None."""
40
+ if not fact.available:
41
+ return f"n/a ({fact.reason})"
42
+ if fact.value is None:
43
+ return none_text
44
+ return fmt(fact.value) if fmt is not None else str(fact.value)
@@ -0,0 +1,138 @@
1
+ """Terminal rendering. Formatting only; no Spark import."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from typing import TYPE_CHECKING, Any, TypeVar
7
+
8
+ from dataowl.render.format import format_bytes, format_datetime, format_fact, format_int
9
+
10
+ if TYPE_CHECKING:
11
+ from dataowl.model.facts import Fact
12
+ from dataowl.model.overview import ColumnInfo, Overview
13
+
14
+ T = TypeVar("T")
15
+
16
+ NOT_SET = "not set (default)"
17
+ _COLUMN_COMMENT_WIDTH = 40
18
+ _TABLE_COMMENT_WIDTH = 80
19
+
20
+
21
+ def render_overview(overview: Overview) -> str:
22
+ header = (
23
+ f"{overview.table.display_name()} "
24
+ f"({_header_value(overview.object_type_raw)}, {_header_value(overview.format)})"
25
+ )
26
+ rows = [
27
+ ("Size:", format_fact(overview.size_bytes, format_bytes)),
28
+ ("Files:", _files(overview)),
29
+ ("Rows:", format_fact(overview.num_rows, format_int)),
30
+ ("Columns:", _columns_count(overview)),
31
+ ("Partitioned by:", format_fact(overview.partition_columns, _names)),
32
+ ("Clustered by:", format_fact(overview.clustering_columns, _names)),
33
+ ("Change Data Feed:", format_fact(overview.change_data_feed, none_text=NOT_SET)),
34
+ ("Log retention:", format_fact(overview.log_retention, none_text=NOT_SET)),
35
+ (
36
+ "Deleted file retention:",
37
+ format_fact(overview.deleted_file_retention, none_text=NOT_SET),
38
+ ),
39
+ ("Created:", format_fact(overview.created, format_datetime)),
40
+ ("Last modified:", format_fact(overview.last_modified, format_datetime)),
41
+ ("Owner:", format_fact(overview.owner)),
42
+ ("Comment:", format_fact(overview.comment, _table_comment)),
43
+ ]
44
+ width = max(len(label) for label, _ in rows) + 2
45
+ lines = [header, ""]
46
+ lines += [f"{label:<{width}}{value}" for label, value in rows]
47
+ lines += ["", "SCHEMA"]
48
+ lines += _schema(overview.columns)
49
+ return "\n".join(lines)
50
+
51
+
52
+ def _files(overview: Overview) -> str:
53
+ return _with_secondary(
54
+ format_fact(overview.num_files, format_int),
55
+ overview.num_files,
56
+ overview.avg_file_size_bytes,
57
+ format_bytes,
58
+ "avg {}",
59
+ "avg",
60
+ )
61
+
62
+
63
+ def _columns_count(overview: Overview) -> str:
64
+ return _with_secondary(
65
+ format_fact(overview.num_columns, format_int),
66
+ overview.num_columns,
67
+ overview.num_fields_nested,
68
+ format_int,
69
+ "{} incl. nested",
70
+ "incl. nested",
71
+ )
72
+
73
+
74
+ def _with_secondary(
75
+ text: str,
76
+ primary: Fact[Any],
77
+ secondary: Fact[T],
78
+ fmt: Callable[[T], str],
79
+ template: str,
80
+ label: str,
81
+ ) -> str:
82
+ """Append ' (<secondary>)' to text.
83
+
84
+ The secondary part is left out when both facts are unavailable for the same reason.
85
+ """
86
+ if not secondary.available and secondary.reason == primary.reason:
87
+ return text
88
+ if not secondary.available or secondary.value is None:
89
+ return f"{text} ({label}: {format_fact(secondary)})"
90
+ return f"{text} ({template.format(fmt(secondary.value))})"
91
+
92
+
93
+ def _names(names: tuple[str, ...]) -> str:
94
+ return ", ".join(names) if names else "–"
95
+
96
+
97
+ def _schema(columns: Fact[tuple[ColumnInfo, ...]]) -> list[str]:
98
+ if not columns.available or columns.value is None:
99
+ return [f" {format_fact(columns)}"]
100
+
101
+ table = [("#", "name", "type", "nullable", "comment")]
102
+ table += [
103
+ (
104
+ str(number),
105
+ column.name,
106
+ column.data_type,
107
+ "yes" if column.nullable else "no",
108
+ _shorten(column.comment or "", _COLUMN_COMMENT_WIDTH),
109
+ )
110
+ for number, column in enumerate(columns.value, start=1)
111
+ ]
112
+ widths = [max(len(row[i]) for row in table) for i in range(4)]
113
+ return [
114
+ (
115
+ " "
116
+ + " ".join(cell.ljust(width) for cell, width in zip(row[:4], widths, strict=True))
117
+ + " "
118
+ + row[4]
119
+ ).rstrip()
120
+ for row in table
121
+ ]
122
+
123
+
124
+ def _header_value(fact: Fact[str]) -> str:
125
+ """Header value; the reason for an unavailable fact is shown on the lines below."""
126
+ return format_fact(fact) if fact.available else "n/a"
127
+
128
+
129
+ def _table_comment(comment: str) -> str:
130
+ return _shorten(comment, _TABLE_COMMENT_WIDTH) or "–"
131
+
132
+
133
+ def _shorten(text: str, width: int) -> str:
134
+ """Collapse line breaks and whitespace, and cut to width characters including '…'."""
135
+ text = " ".join(text.split())
136
+ if len(text) > width:
137
+ return text[: width - 1] + "…"
138
+ return text
dataowl/runner.py ADDED
@@ -0,0 +1,47 @@
1
+ """SqlRunner protocol and SparkRunner.
2
+
3
+ PySpark is imported lazily, so this module can be imported without PySpark installed.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from typing import TYPE_CHECKING, Any, Protocol
9
+
10
+ from dataowl.identifiers import TableRef
11
+
12
+ if TYPE_CHECKING:
13
+ from pyspark.sql import SparkSession
14
+ from pyspark.sql.types import StructType
15
+
16
+
17
+ class SqlRunner(Protocol):
18
+ def query(self, sql: str, params: dict[str, Any] | None = None) -> list[dict[str, Any]]: ...
19
+
20
+ def schema(self, table: TableRef) -> StructType: ...
21
+
22
+
23
+ class SparkRunner:
24
+ def __init__(self, spark: SparkSession) -> None:
25
+ self._spark = spark
26
+
27
+ def query(self, sql: str, params: dict[str, Any] | None = None) -> list[dict[str, Any]]:
28
+ rows = self._spark.sql(sql, args=params).collect()
29
+ return [row.asDict(recursive=True) for row in rows]
30
+
31
+ def schema(self, table: TableRef) -> StructType:
32
+ # spark.table() is lazy; reading .schema does not scan data.
33
+ return self._spark.table(table.quoted()).schema
34
+
35
+
36
+ def get_runner(spark: SparkSession | None = None) -> SparkRunner:
37
+ """Return a SparkRunner for the given session, or for the active session."""
38
+ if spark is None:
39
+ from pyspark.sql import SparkSession
40
+
41
+ spark = SparkSession.getActiveSession()
42
+ if spark is None:
43
+ raise RuntimeError(
44
+ "No active SparkSession found. Run dataowl in a Databricks notebook or job, "
45
+ "or pass a session explicitly with spark=."
46
+ )
47
+ return SparkRunner(spark)
@@ -0,0 +1,158 @@
1
+ Metadata-Version: 2.5
2
+ Name: dataowl
3
+ Version: 0.1.0
4
+ Summary: Facts about Databricks data products for building dbt staging models.
5
+ Project-URL: Homepage, https://github.com/alexD1990/dataowl
6
+ Project-URL: Repository, https://github.com/alexD1990/dataowl
7
+ Project-URL: Changelog, https://github.com/alexD1990/dataowl/blob/master/CHANGELOG.md
8
+ Project-URL: Issues, https://github.com/alexD1990/dataowl/issues
9
+ Author: Alexandro Dronnen
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: data-engineering,databricks,dbt,delta-lake,unity-catalog
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3 :: Only
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Database
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.10
24
+ Provides-Extra: dev
25
+ Requires-Dist: mypy==2.3.1; extra == 'dev'
26
+ Requires-Dist: pyspark>=3.4; extra == 'dev'
27
+ Requires-Dist: pytest; extra == 'dev'
28
+ Requires-Dist: ruff==0.16.9; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # dataowl
32
+
33
+ dataowl reports facts about a table in Databricks (Unity Catalog / Delta Lake): size, files,
34
+ rows, schema and table properties. It is meant for data engineers who build a dbt staging
35
+ model and want the facts without writing exploratory SQL. It reports what exists and never
36
+ gives recommendations.
37
+
38
+ ## Requirements
39
+
40
+ - Unity Catalog.
41
+ - Databricks Runtime 14.3 LTS or above, or serverless compute.
42
+ Verified on serverless (Spark Connect).
43
+ - PySpark is provided by Databricks. dataowl has no other runtime dependencies.
44
+
45
+ ## Installation
46
+
47
+ In a Databricks notebook:
48
+
49
+ ```python
50
+ %pip install dataowl==0.1.0
51
+ ```
52
+
53
+ ## Usage
54
+
55
+ ```python
56
+ import dataowl
57
+
58
+ dataowl.inspect("catalog.schema.table").show()
59
+ ```
60
+
61
+ As a JSON-serializable dict, where every fact has a value, a source and a reason when it
62
+ is unavailable:
63
+
64
+ ```python
65
+ overview = dataowl.inspect("catalog.schema.table")
66
+ overview.to_dict()
67
+ ```
68
+
69
+ Row counts are skipped for views, materialized views and foreign tables unless you ask for
70
+ them:
71
+
72
+ ```python
73
+ dataowl.inspect("catalog.schema.some_view", count_views=True).show()
74
+ ```
75
+
76
+ Names with characters other than letters, digits and underscore are quoted with backticks:
77
+ `` dataowl.inspect("`my-catalog`.schema.table") ``.
78
+
79
+ ## Example output
80
+
81
+ Output for Databricks' public sample data (`samples.nyctaxi.trips`):
82
+
83
+ ```text
84
+ samples.nyctaxi.trips (MANAGED, DELTA)
85
+
86
+ Size: 354.2 KB
87
+ Files: 1 (avg 354.2 KB)
88
+ Rows: 21 932
89
+ Columns: 6 (6 incl. nested)
90
+ Partitioned by: –
91
+ Clustered by: –
92
+ Change Data Feed: true
93
+ Log retention: not set (default)
94
+ Deleted file retention: not set (default)
95
+ Created: 2025-09-30 11:28
96
+ Last modified: 2026-09-14 15:07
97
+ Owner: System user
98
+ Comment: –
99
+
100
+ SCHEMA
101
+ # name type nullable comment
102
+ 1 tpep_pickup_datetime timestamp yes
103
+ 2 tpep_dropoff_datetime timestamp yes
104
+ 3 trip_distance double yes
105
+ 4 fare_amount double yes
106
+ 5 pickup_zip int yes
107
+ 6 dropoff_zip int yes
108
+ ```
109
+
110
+ ## Principles
111
+
112
+ - **Read-only.** Only `SELECT`, `DESCRIBE` and `SHOW`. Nothing is written, optimized or
113
+ analyzed.
114
+ - **Facts only.** No recommendations or assessments.
115
+ - **No row values leave Spark.** Only aggregates and metadata are collected.
116
+ - **Unavailable facts are reported with a reason.** Missing permissions or an unsupported
117
+ object type make the affected facts `n/a (<reason>)`; the rest of the report is still
118
+ produced. Only a table that does not exist or is not accessible raises
119
+ `TableNotFoundError`.
120
+
121
+ ## Facts per object type
122
+
123
+ | Facts | Source | MANAGED, EXTERNAL | STREAMING_TABLE, MATERIALIZED_VIEW | FOREIGN | VIEW | UNKNOWN or n/a type |
124
+ |---|---|---|---|---|---|---|
125
+ | Object type, format, owner, comment, created | `information_schema.tables` | yes | yes | yes | yes | yes (n/a if the query fails) |
126
+ | Size, files, avg file size, last modified, partitioning, clustering | `DESCRIBE DETAIL` | yes | attempted¹ | attempted¹ | no | attempted¹ |
127
+ | Columns and schema | `information_schema.columns` | yes | yes | yes | yes | yes |
128
+ | Field count incl. nested | Spark schema | yes | yes | yes | yes | yes |
129
+ | Change Data Feed, log and deleted file retention | `SHOW TBLPROPERTIES` | yes | attempted¹ | attempted¹ | no | attempted¹ |
130
+ | Rows | `COUNT(*)` | yes | MATERIALIZED_VIEW: with `count_views=True`; STREAMING_TABLE: yes | with `count_views=True` | with `count_views=True` | UNKNOWN: yes; n/a type: with `count_views=True` |
131
+
132
+ ¹ The query runs. If Databricks does not support it for the object, the facts are shown as
133
+ `n/a (<reason>)`.
134
+
135
+ "UNKNOWN" is a `table_type` dataowl does not recognize, such as `MANAGED_SHALLOW_CLONE`. The
136
+ original value is shown in the header. "n/a type" means the object type could not be read.
137
+
138
+ ## Fields
139
+
140
+ - **Last modified** is the last data change, from `DESCRIBE DETAIL`. **Created** is from
141
+ `information_schema.tables`.
142
+ - **Rows** is an exact `COUNT(*)`. On large tables this costs compute.
143
+ - **Files (avg ...)** is the size divided by the number of files.
144
+ - **Columns (... incl. nested)** counts every field, including fields inside structs, array
145
+ elements and map keys and values.
146
+ - **not set (default)** means the table property is not set, so Databricks uses its default.
147
+ dataowl does not report what the default is.
148
+ - **#** in the schema table is a running number from 1 in column order.
149
+
150
+ ## Known deviations
151
+
152
+ - `information_schema.columns.ordinal_position` is documented as numbered from 1, but has
153
+ been observed 0-based in Databricks (serverless, October 2026). dataowl keeps the raw
154
+ value in `ColumnInfo.position` and shows a running number in the `#` column.
155
+
156
+ ## License
157
+
158
+ MIT. Copyright (c) 2026 Alexandro Dronnen. See [LICENSE](LICENSE).
@@ -0,0 +1,26 @@
1
+ dataowl/__init__.py,sha256=1w3A_hT9E3P7__pJvcbl_zWXhdUPk7zwLDaOyfKzXBU,2320
2
+ dataowl/errors.py,sha256=E4zTLIA-W8DwS8gw77qKGSEVFC6iIGNVfD1Xg3eC_8w,188
3
+ dataowl/identifiers.py,sha256=kWF0UR0OYukQrfW0hLqlEreFeWlMWnIZVt7-OSP9oZg,3600
4
+ dataowl/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ dataowl/runner.py,sha256=94hsS2sZCj3HZIxkE0qstJOYhR2MAufkuKgriHbVlyw,1541
6
+ dataowl/collect/__init__.py,sha256=X3PmsdPlylbxQpOdz2uccoTb98yA3Nzy-uBKq1il9eg,1248
7
+ dataowl/collect/columns.py,sha256=_tO4kp58gxZ9h6cnkZgYaI9EXFDadGIANbGrVLj8yjM,4899
8
+ dataowl/collect/counts.py,sha256=2EyvOjKBTOJV7ecSdkuuHazkJRi2YTIEkD5YUxISt2s,1635
9
+ dataowl/collect/detail.py,sha256=QYKHnt6xfQXTVSnfIInMm3h9Xsh1hzOSx-CPN7A8Wu4,2987
10
+ dataowl/collect/history.py,sha256=U4S_2y3zgLZVfMenHRaJFBW8yqh2mUBuI291LGQVOJ8,35
11
+ dataowl/collect/keys.py,sha256=U4S_2y3zgLZVfMenHRaJFBW8yqh2mUBuI291LGQVOJ8,35
12
+ dataowl/collect/properties.py,sha256=-RcrYz1AZLGf3waO7s_gQQuwR1rmex20pQEY2AfTMEs,2766
13
+ dataowl/collect/tables.py,sha256=FmbwGxXfhZ6KuJTUuQcPt8x5as9pXASi17mJaJbt7Do,2698
14
+ dataowl/collect/timestamps.py,sha256=U4S_2y3zgLZVfMenHRaJFBW8yqh2mUBuI291LGQVOJ8,35
15
+ dataowl/model/__init__.py,sha256=xpc0GSN5DNctiUqc0s5lBSvpfoCxAZ7ly119pctfUT4,89
16
+ dataowl/model/analysis.py,sha256=U4S_2y3zgLZVfMenHRaJFBW8yqh2mUBuI291LGQVOJ8,35
17
+ dataowl/model/facts.py,sha256=8mB5Y43KC3lm-K-BHQsyLGI072rM9-afHAhNXK1ShNg,2723
18
+ dataowl/model/history.py,sha256=U4S_2y3zgLZVfMenHRaJFBW8yqh2mUBuI291LGQVOJ8,35
19
+ dataowl/model/overview.py,sha256=FQwgu6LGhHsTXCYL6iKfZtfluzBbHCz7-oI5zZNwod8,3544
20
+ dataowl/render/__init__.py,sha256=HBBTqO_j52tlMDbPWgOiNY1yPCGpaLj2hdFmyRMMKzk,78
21
+ dataowl/render/format.py,sha256=WDwKrI0LjauuY1RAxtduReTni5-UpMmyEppfV4Ar_yc,1312
22
+ dataowl/render/terminal.py,sha256=iIQEAz_28IX0zy8dCK-QknAE9bDLb4qqPaeH9wLAif0,4470
23
+ dataowl-0.1.0.dist-info/METADATA,sha256=marLaf09bCi5ViXqjhlqvUf2RVGZIYABFXWQmyzzcw0,6131
24
+ dataowl-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
25
+ dataowl-0.1.0.dist-info/licenses/LICENSE,sha256=bO2mV8F0lNY78kr-gUhW-EA7YI9uFmOXD3KmktyNjlI,1074
26
+ dataowl-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Alexandro Dronnen
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.