dash-control 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,103 @@
1
+ Metadata-Version: 2.4
2
+ Name: dash-control
3
+ Version: 0.1.1
4
+ Summary: Databricks Control Center — live system-table dashboard for platform, governance, cost, users, jobs, and queries
5
+ Project-URL: Homepage, https://github.com/dash-libs/dash-control
6
+ Author-email: Darshan Shah <darshan.innovation@gmail.com>
7
+ License: Apache-2.0
8
+ Keywords: control-center,cost,dashboard,databricks,governance,observability,system-tables,unity-catalog
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Intended Audience :: Developers
11
+ Classifier: Intended Audience :: End Users/Desktop
12
+ Classifier: Intended Audience :: Information Technology
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: Apache Software License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Requires-Python: >=3.9
21
+ Requires-Dist: dash-uis==0.2.3
22
+ Requires-Dist: ipywidgets>=8.0
23
+ Provides-Extra: dev
24
+ Requires-Dist: hatch; extra == 'dev'
25
+ Requires-Dist: pdoc; extra == 'dev'
26
+ Requires-Dist: pytest; extra == 'dev'
27
+ Requires-Dist: pytest-cov; extra == 'dev'
28
+ Requires-Dist: ruff; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # dash-control
32
+
33
+ **Databricks Control Center** — a live, 8-panel system-table dashboard for platform, governance, cost, and operations teams. One `%pip install` + one `launch()` call gives you an always-on view of your entire Databricks workspace.
34
+
35
+ ```python
36
+ %pip install dash-control
37
+ from dashcontrol import launch, ControlCenterConfig
38
+
39
+ config = ControlCenterConfig(
40
+ date_range_days=30,
41
+ catalogs=["main", "analytics"],
42
+ )
43
+ launch(config)
44
+ ```
45
+
46
+ ## Panels
47
+
48
+ | Panel | What it shows |
49
+ |---|---|
50
+ | **Health** | DBU today, active users, failed jobs (last 24 h), total table count, 7-day DBU sparkline |
51
+ | **Cost** | Daily spend by SKU, top clusters, top jobs, top users by DBU, burn-rate projection |
52
+ | **Users** | Top users by query count & tables accessed, inactive users, permission changes |
53
+ | **Catalog** | Table inventory, tables by schema, stale tables, column count distribution, most-accessed tables |
54
+ | **Jobs** | Success rate, top failures, longest runs, daily run volume trend |
55
+ | **Queries** | Slowest queries, most expensive queries, top users, error summary |
56
+ | **Governance** | Tables without owners, PII columns, access anomalies, schema changes |
57
+ | **Custom** | Run any SQL against system tables; define custom panels in `ControlCenterConfig` |
58
+
59
+ ## Configuration
60
+
61
+ ```python
62
+ from dashcontrol import ControlCenterConfig, CustomPanel
63
+
64
+ config = ControlCenterConfig(
65
+ date_range_days=14, # default look-back window (also adjustable in the UI)
66
+ catalogs=["main"], # limit catalog scope; empty = all catalogs
67
+ panels=["health", "cost"], # show only these panels
68
+ row_limit=200, # max rows per table result
69
+ workspace_name="prod", # displayed in the header
70
+ custom_panels=[
71
+ CustomPanel(
72
+ name="My Query",
73
+ sql="SELECT user_name, COUNT(*) AS cnt FROM system.access.audit GROUP BY 1",
74
+ description="Custom audit rollup",
75
+ )
76
+ ],
77
+ )
78
+ ```
79
+
80
+ ## Requirements
81
+
82
+ - Databricks Runtime 13.3 LTS or later (system tables must be enabled)
83
+ - Unity Catalog workspace
84
+ - `ipywidgets` (included as a dependency)
85
+
86
+ ## Architecture
87
+
88
+ Every panel is **lazy-loaded** — clicking Load runs the query; nothing executes on `launch()`. Queries target Databricks system tables (`system.billing.*`, `system.access.audit`, `system.jobs.*`, `system.query.history`, `system.information_schema.*`). All queries degrade gracefully: if a system table isn't available on your workspace tier, the panel shows an info banner instead of crashing.
89
+
90
+ ## Part of the DashLibs suite
91
+
92
+ | Package | Purpose |
93
+ |---|---|
94
+ | [dash-dq](https://pypi.org/project/dash-dq/) | 60+ data quality checks |
95
+ | [dash-synthetic](https://pypi.org/project/dash-synthetic/) | Synthetic data generation |
96
+ | [dash-observe](https://pypi.org/project/dash-observe/) | Freshness, volume & schema monitoring |
97
+ | [dash-gov](https://pypi.org/project/dash-gov/) | Table/column lineage + role classification |
98
+ | [dash-ontology](https://pypi.org/project/dash-ontology/) | Auto-inferred business ontology from lineage |
99
+ | **dash-control** | Control Center — this package |
100
+
101
+ ## License
102
+
103
+ Apache 2.0
@@ -0,0 +1,9 @@
1
+ dashcontrol/__init__.py,sha256=1k2hWujkfIk-F8Nm5yYxWDd2NRjEWGD6JdMZQA3SORg,229
2
+ dashcontrol/config.py,sha256=pd8OoxDMjHtzSC_8iGrICd76inlMzJl6lDV71jTvbeA,2152
3
+ dashcontrol/formatters.py,sha256=pF81n5HivSK2pXYQn5ztZxhgBpTGg288T2ZAORMdJvk,5528
4
+ dashcontrol/runner.py,sha256=u05oNBdqKORmtQSobkW6Ln-7DZ92db9L6GKbu08H3dU,2511
5
+ dashcontrol/sql.py,sha256=6qCVek5EvvN7iGYPVnb5j2EU2GQhzE6YDVDJ4B0y9cg,18083
6
+ dashcontrol/ui.py,sha256=ENWTFrEZ1gchWgt4OheC4CdSslc3Rzx6kYqmJWC19qs,20934
7
+ dash_control-0.1.1.dist-info/METADATA,sha256=7tdps6aVCPdM6t13lUQP7tE29Ab8-ieCMSHzzdYopqM,4457
8
+ dash_control-0.1.1.dist-info/WHEEL,sha256=mffPy8wBnZQn2VnJUU5jE99KsxaSfiyMHV9Yt0aLVxs,87
9
+ dash_control-0.1.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.30.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,6 @@
1
+ """DashControl — Databricks Control Center."""
2
+ from dashcontrol.config import ControlCenterConfig, CustomPanel
3
+ from dashcontrol.ui import launch
4
+
5
+ __version__ = "0.1.1"
6
+ __all__ = ["ControlCenterConfig", "CustomPanel", "launch"]
dashcontrol/config.py ADDED
@@ -0,0 +1,62 @@
1
+ """
2
+ ControlCenterConfig — user-facing configuration for the Control Center.
3
+
4
+ All fields have sensible defaults so teams can call launch() with zero config,
5
+ or pre-wire a domain-specific dashboard by passing a config object.
6
+ """
7
+ from __future__ import annotations
8
+ from dataclasses import dataclass, field
9
+
10
+ ALL_PANELS = ["health", "cost", "users", "catalog", "jobs", "queries", "governance"]
11
+
12
+
13
+ @dataclass
14
+ class CustomPanel:
15
+ title: str
16
+ sql: str
17
+ description: str = ""
18
+
19
+
20
+ @dataclass
21
+ class ControlCenterConfig:
22
+ """
23
+ Configuration for the Databricks Control Center.
24
+
25
+ Parameters
26
+ ----------
27
+ date_range_days
28
+ Default lookback window for all time-based queries (default: 30).
29
+ catalogs
30
+ List of catalogs to scope table/lineage queries to.
31
+ Empty list means all visible catalogs.
32
+ panels
33
+ Which built-in panels to show. Defaults to all.
34
+ custom_panels
35
+ User-defined SQL panels appended after the built-in tabs.
36
+ row_limit
37
+ Max rows returned per panel query (default: 500).
38
+ workspace_name
39
+ Optional display name shown in the dashboard header.
40
+ """
41
+ date_range_days: int = 30
42
+ catalogs: list[str] = field(default_factory=list)
43
+ panels: list[str] = field(default_factory=lambda: list(ALL_PANELS))
44
+ custom_panels: list[CustomPanel] = field(default_factory=list)
45
+ row_limit: int = 500
46
+ workspace_name: str = ""
47
+
48
+ def __post_init__(self):
49
+ unknown = set(self.panels) - set(ALL_PANELS)
50
+ if unknown:
51
+ raise ValueError(f"Unknown panels: {unknown}. Valid: {ALL_PANELS}")
52
+ if self.date_range_days < 1 or self.date_range_days > 365:
53
+ raise ValueError("date_range_days must be between 1 and 365")
54
+ if self.row_limit < 1 or self.row_limit > 10_000:
55
+ raise ValueError("row_limit must be between 1 and 10,000")
56
+
57
+ def catalog_filter(self, col: str = "table_catalog") -> str:
58
+ """SQL WHERE fragment to filter by configured catalogs."""
59
+ if not self.catalogs:
60
+ return ""
61
+ quoted = ", ".join(f"'{c}'" for c in self.catalogs)
62
+ return f"AND {col} IN ({quoted})"
@@ -0,0 +1,161 @@
1
+ """
2
+ Pure-Python formatting utilities — no Spark dependency.
3
+
4
+ Converts query result rows (list[dict]) into styled HTML tables,
5
+ stat tiles, trend sparklines, and CSV exports for the Control Center UI.
6
+ """
7
+ from __future__ import annotations
8
+ from typing import Optional
9
+
10
+ # Colour palette
11
+ _TEAL = "#2A9D90"
12
+ _AMBER = "#F4A261"
13
+ _RED = "#E63946"
14
+ _GREEN = "#2DC653"
15
+ _GREY = "#6B7280"
16
+ _LIGHT = "#F9FAFB"
17
+ _BORDER = "#E5E7EB"
18
+
19
+
20
+ def stat_tile(label: str, value, color: str = _TEAL, unit: str = "") -> str:
21
+ """Render a single KPI tile (label + big number)."""
22
+ return (
23
+ f"<div style='display:inline-block;padding:14px 20px;margin:6px;"
24
+ f"border-radius:8px;background:{_LIGHT};border:1px solid {_BORDER};"
25
+ f"min-width:140px;text-align:center'>"
26
+ f"<div style='font-size:26px;font-weight:700;color:{color}'>"
27
+ f"{value}{unit}</div>"
28
+ f"<div style='font-size:11px;color:{_GREY};margin-top:4px'>{label}</div>"
29
+ f"</div>"
30
+ )
31
+
32
+
33
+ def stat_row(tiles: list[str]) -> str:
34
+ """Wrap stat tiles in a flex row."""
35
+ inner = "".join(tiles)
36
+ return f"<div style='display:flex;flex-wrap:wrap;gap:4px;margin-bottom:12px'>{inner}</div>"
37
+
38
+
39
+ def html_table(
40
+ rows: list[dict],
41
+ highlight_col: Optional[str] = None,
42
+ max_rows: int = 200,
43
+ col_widths: Optional[dict] = None,
44
+ ) -> str:
45
+ """
46
+ Convert a list of row-dicts to a styled HTML table.
47
+
48
+ highlight_col — column whose values determine row background colour
49
+ (high numeric value = amber, error strings = red)
50
+ """
51
+ if not rows:
52
+ return "<div style='color:#9ca3af;font-size:12px;padding:8px'>No data</div>"
53
+
54
+ cols = list(rows[0].keys())
55
+ col_widths = col_widths or {}
56
+
57
+ def _header(col: str) -> str:
58
+ w = f"min-width:{col_widths[col]};" if col in col_widths else ""
59
+ label = col.replace("_", " ").title()
60
+ return (
61
+ f"<th style='padding:6px 10px;background:#F3F4F6;text-align:left;"
62
+ f"font-size:11px;color:{_GREY};font-weight:600;white-space:nowrap;{w}'>"
63
+ f"{label}</th>"
64
+ )
65
+
66
+ def _cell(val, col: str) -> str:
67
+ text = "" if val is None else str(val)
68
+ # Truncate long strings
69
+ display = text[:80] + "…" if len(text) > 80 else text
70
+ return (
71
+ f"<td style='padding:5px 10px;font-size:12px;"
72
+ f"font-family:monospace;border-top:1px solid {_BORDER}'>"
73
+ f"{display}</td>"
74
+ )
75
+
76
+ def _row_bg(row: dict) -> str:
77
+ if highlight_col and highlight_col in row:
78
+ v = row[highlight_col]
79
+ if isinstance(v, str) and any(k in v.upper() for k in ("FAIL", "ERROR", "DENIED")):
80
+ return "background:#FEF2F2;"
81
+ return ""
82
+
83
+ headers = "".join(_header(c) for c in cols)
84
+ body_rows = ""
85
+ for r in rows[:max_rows]:
86
+ bg = _row_bg(r)
87
+ cells = "".join(_cell(r.get(c), c) for c in cols)
88
+ body_rows += f"<tr style='{bg}'>{cells}</tr>"
89
+
90
+ footer = ""
91
+ if len(rows) > max_rows:
92
+ footer = (
93
+ f"<tr><td colspan='{len(cols)}' style='padding:6px 10px;"
94
+ f"font-size:11px;color:{_GREY};text-align:center'>"
95
+ f"Showing {max_rows} of {len(rows)} rows</td></tr>"
96
+ )
97
+
98
+ return (
99
+ f"<div style='overflow-x:auto'>"
100
+ f"<table style='border-collapse:collapse;width:100%;font-size:12px'>"
101
+ f"<thead><tr>{headers}</tr></thead>"
102
+ f"<tbody>{body_rows}{footer}</tbody>"
103
+ f"</table></div>"
104
+ )
105
+
106
+
107
+ def error_box(message: str) -> str:
108
+ """Red alert box for query errors."""
109
+ return (
110
+ f"<div style='padding:10px 14px;background:#FEF2F2;border:1px solid #FCA5A5;"
111
+ f"border-radius:6px;color:{_RED};font-size:12px;font-family:monospace'>"
112
+ f"⚠ {message}</div>"
113
+ )
114
+
115
+
116
+ def info_box(message: str) -> str:
117
+ """Blue info box."""
118
+ return (
119
+ f"<div style='padding:10px 14px;background:#EFF6FF;border:1px solid #BFDBFE;"
120
+ f"border-radius:6px;color:#1D4ED8;font-size:12px'>"
121
+ f"ℹ {message}</div>"
122
+ )
123
+
124
+
125
+ def section_header(title: str, subtitle: str = "") -> str:
126
+ sub = f"<span style='font-size:11px;color:{_GREY};margin-left:8px'>{subtitle}</span>" if subtitle else ""
127
+ return (
128
+ f"<div style='font-weight:600;font-size:13px;color:#374151;"
129
+ f"margin:12px 0 6px;padding-bottom:4px;border-bottom:1px solid {_BORDER}'>"
130
+ f"{title}{sub}</div>"
131
+ )
132
+
133
+
134
+ def sparkline_html(values: list[float], label: str = "") -> str:
135
+ """
136
+ Minimal ASCII sparkline for trend data (no JS/SVG required).
137
+ Uses Unicode block characters: ▁▂▃▄▅▆▇█
138
+ """
139
+ blocks = "▁▂▃▄▅▆▇█"
140
+ if not values or all(v == 0 for v in values):
141
+ return f"<span style='font-family:monospace;color:{_GREY}'>{'▁' * 10}</span>"
142
+ mn, mx = min(values), max(values)
143
+ rng = mx - mn or 1
144
+ chars = "".join(blocks[min(7, int((v - mn) / rng * 7))] for v in values)
145
+ return (
146
+ f"<span style='font-family:monospace;font-size:14px;color:{_TEAL}'>{chars}</span>"
147
+ + (f"<span style='font-size:10px;color:{_GREY};margin-left:4px'>{label}</span>" if label else "")
148
+ )
149
+
150
+
151
+ def format_number(n) -> str:
152
+ """Format large numbers with K/M suffixes."""
153
+ try:
154
+ n = float(n)
155
+ except (TypeError, ValueError):
156
+ return str(n)
157
+ if abs(n) >= 1_000_000:
158
+ return f"{n / 1_000_000:.1f}M"
159
+ if abs(n) >= 1_000:
160
+ return f"{n / 1_000:.1f}K"
161
+ return f"{n:,.0f}" if n == int(n) else f"{n:,.2f}"
dashcontrol/runner.py ADDED
@@ -0,0 +1,78 @@
1
+ """
2
+ Query runner — executes SQL against Databricks system tables.
3
+
4
+ Spark is imported inside functions only (never at module level) so this
5
+ module can be imported and tested without a live Spark session.
6
+ """
7
+ from __future__ import annotations
8
+ from typing import Optional
9
+
10
+
11
+ class QueryResult:
12
+ """Holds the result of a panel query with metadata."""
13
+
14
+ def __init__(self, rows: list[dict], sql: str, error: Optional[str] = None):
15
+ self.rows = rows
16
+ self.sql = sql
17
+ self.error = error
18
+ self.ok = error is None
19
+
20
+ def __len__(self) -> int:
21
+ return len(self.rows)
22
+
23
+ def column_names(self) -> list[str]:
24
+ return list(self.rows[0].keys()) if self.rows else []
25
+
26
+ def column_values(self, col: str) -> list:
27
+ return [r.get(col) for r in self.rows]
28
+
29
+ def first_value(self, col: str, default=None):
30
+ return self.rows[0].get(col, default) if self.rows else default
31
+
32
+ def to_csv(self) -> str:
33
+ if not self.rows:
34
+ return ""
35
+ cols = self.column_names()
36
+ lines = [",".join(str(c) for c in cols)]
37
+ for row in self.rows:
38
+ lines.append(",".join(str(row.get(c, "")) for c in cols))
39
+ return "\n".join(lines)
40
+
41
+
42
+ def run_query(sql: str, limit: int = 500) -> QueryResult:
43
+ """
44
+ Execute a SQL query against Databricks system tables.
45
+
46
+ Uses the active SparkSession — must be run inside a Databricks notebook
47
+ (or any environment where a SparkSession is already active).
48
+
49
+ Parameters
50
+ ----------
51
+ sql : SQL string (use functions from dashcontrol.sql)
52
+ limit : max rows to return
53
+
54
+ Returns
55
+ -------
56
+ QueryResult with .rows (list of dicts) and .ok / .error
57
+ """
58
+ try:
59
+ from pyspark.sql import SparkSession
60
+ spark = SparkSession.getActiveSession()
61
+ if spark is None:
62
+ return QueryResult([], sql, error="No active Spark session. Run inside a Databricks notebook.")
63
+ df = spark.sql(sql.strip())
64
+ rows = [r.asDict() for r in df.limit(limit).collect()]
65
+ return QueryResult(rows, sql)
66
+ except Exception as e:
67
+ return QueryResult([], sql, error=str(e))
68
+
69
+
70
+ def run_query_safe(sql: str, limit: int = 500) -> QueryResult:
71
+ """
72
+ Same as run_query but returns a QueryResult with error set (never raises)
73
+ even for system tables that don't exist on the current workspace tier.
74
+ """
75
+ try:
76
+ return run_query(sql, limit)
77
+ except Exception as e:
78
+ return QueryResult([], sql, error=str(e))