chartwright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- chartwright/__init__.py +7 -0
- chartwright/absorb.py +93 -0
- chartwright/apply.py +409 -0
- chartwright/cli.py +222 -0
- chartwright/client.py +227 -0
- chartwright/compiler.py +601 -0
- chartwright/dashdiff.py +208 -0
- chartwright/decompile.py +625 -0
- chartwright/ids.py +21 -0
- chartwright/mcp_server.py +112 -0
- chartwright/profiles.py +143 -0
- chartwright/resolver.py +196 -0
- chartwright/sketch.py +180 -0
- chartwright/smoke.py +124 -0
- chartwright/spec.py +597 -0
- chartwright/testing.py +31 -0
- chartwright-0.1.0.dist-info/METADATA +189 -0
- chartwright-0.1.0.dist-info/RECORD +23 -0
- chartwright-0.1.0.dist-info/WHEEL +5 -0
- chartwright-0.1.0.dist-info/entry_points.txt +3 -0
- chartwright-0.1.0.dist-info/licenses/LICENSE +202 -0
- chartwright-0.1.0.dist-info/licenses/NOTICE +5 -0
- chartwright-0.1.0.dist-info/top_level.txt +1 -0
chartwright/decompile.py
ADDED
|
@@ -0,0 +1,625 @@
|
|
|
1
|
+
"""Decompiler: Superset export bundle -> spec (+ explicit lossiness report).
|
|
2
|
+
|
|
3
|
+
Turns any existing dashboard into a diffable, editable spec. Decompilation is
|
|
4
|
+
honest about loss: everything outside the spec surface is NAMED in the loss
|
|
5
|
+
report, never silently dropped-and-forgotten.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import io
|
|
11
|
+
import re
|
|
12
|
+
import zipfile
|
|
13
|
+
from dataclasses import asdict, dataclass, field
|
|
14
|
+
from typing import Callable, get_args
|
|
15
|
+
|
|
16
|
+
import yaml
|
|
17
|
+
|
|
18
|
+
from .compiler import ROW_UNITS_PER_SPEC_UNIT, SDC_BAR_MARKER, VIZ_TYPE
|
|
19
|
+
from .spec import ADHOC_AGGREGATES, FORMAT_COLOR_HEX, FilterOp
|
|
20
|
+
|
|
21
|
+
REVERSE_VIZ = {v: k for k, v in VIZ_TYPE.items() if k != "bar"} # echarts_timeseries_bar -> timeseries_bar
|
|
22
|
+
_FILTER_OPS = set(get_args(FilterOp))
|
|
23
|
+
|
|
24
|
+
# Cosmetic / behavioral params we knowingly discard without a loss entry.
|
|
25
|
+
_IGNORABLE = {
|
|
26
|
+
"datasource", "viz_type", "adhoc_filters", "extra_form_data", "dashboards",
|
|
27
|
+
"color_scheme", "legendType", "legendOrientation", "legendMargin", "show_legend",
|
|
28
|
+
"rich_tooltip", "annotation_layers", "comparison_type", "forecastEnabled",
|
|
29
|
+
"forecastInterval", "forecastPeriods", "forecastSeasonalityDaily",
|
|
30
|
+
"forecastSeasonalityWeekly", "forecastSeasonalityYearly", "truncateXAxis",
|
|
31
|
+
"truncate_metric", "only_total", "order_desc", "show_empty_columns",
|
|
32
|
+
"sort_series_type", "markerSize", "tooltipTimeFormat", "x_axis_sort_asc",
|
|
33
|
+
"x_axis_sort_series", "x_axis_sort_series_ascending", "x_axis_time_format",
|
|
34
|
+
"x_axis_title_margin", "y_axis_bounds", "y_axis_format", "y_axis_title_margin",
|
|
35
|
+
"y_axis_title_position", "sort_by_metric", "show_labels_threshold",
|
|
36
|
+
"server_page_length", "query_mode", "time_range", "time_grain_sqla", "x_axis",
|
|
37
|
+
"granularity_sqla", "x_axis_sort", "orientation",
|
|
38
|
+
"metric", "metrics", "groupby", "row_limit", "all_columns", "subheader",
|
|
39
|
+
"label_colors", "date_format", "outerRadius", "innerRadius", "donut",
|
|
40
|
+
"show_labels", "labels_outside", "label_type", "number_format", "cache_timeout",
|
|
41
|
+
"order_by_cols", "table_timestamp_format", "show_cell_bars", "include_search",
|
|
42
|
+
"currency_format", "opacity", "seriesType", "show_trend_line",
|
|
43
|
+
"start_y_axis_at_zero", "rolling_type", "header_font_size", "subheader_font_size",
|
|
44
|
+
"time_format", "color_picker", "groupbyRows", "groupbyColumns",
|
|
45
|
+
"aggregateFunction", "metricsLayout", "rowOrder", "colOrder", "valueFormat",
|
|
46
|
+
"temporal_columns_lookup", "normalize_across", "legend_type",
|
|
47
|
+
"linear_color_scheme", "sort_x_axis", "sort_y_axis", "bottom_margin",
|
|
48
|
+
"left_margin", "show_percentage", "show_values", "value_bounds",
|
|
49
|
+
"xscale_interval", "yscale_interval", "column", "bins", "normalize",
|
|
50
|
+
"show_value", "slice_id", "url_params", "percent_calculation_type",
|
|
51
|
+
"show_tooltip_labels", "tooltip_label_type", SDC_BAR_MARKER,
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class Loss:
|
|
57
|
+
where: str
|
|
58
|
+
what: str
|
|
59
|
+
|
|
60
|
+
def as_dict(self) -> dict:
|
|
61
|
+
return asdict(self)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass
|
|
65
|
+
class DecompileResult:
|
|
66
|
+
spec: dict
|
|
67
|
+
losses: list[Loss] = field(default_factory=list)
|
|
68
|
+
dataset_uuids: dict[str, str] = field(default_factory=dict) # chart name -> dataset uuid
|
|
69
|
+
|
|
70
|
+
def losses_json(self) -> list[dict]:
|
|
71
|
+
return [loss.as_dict() for loss in self.losses]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
DatasetLookup = Callable[[str], dict | None]
|
|
75
|
+
"""dataset_uuid -> {'database','schema','table'} or None."""
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _format_to_spec(cf: dict) -> dict | None:
|
|
79
|
+
"""Superset conditional_formatting entry -> FormatRule dict, None if outside surface."""
|
|
80
|
+
color = {v.upper(): k for k, v in FORMAT_COLOR_HEX.items()}.get((cf.get("colorScheme") or "").upper())
|
|
81
|
+
op = cf.get("operator")
|
|
82
|
+
if not color or not cf.get("column") or op not in ("<", ">", "between"):
|
|
83
|
+
return None
|
|
84
|
+
rule: dict = {"metric": cf["column"], "operator": op, "color": color}
|
|
85
|
+
if op == "between":
|
|
86
|
+
rule["target_left"] = cf.get("targetValueLeft")
|
|
87
|
+
rule["target_right"] = cf.get("targetValueRight")
|
|
88
|
+
if rule["target_left"] is None or rule["target_right"] is None:
|
|
89
|
+
return None
|
|
90
|
+
else:
|
|
91
|
+
rule["target"] = cf.get("targetValue")
|
|
92
|
+
if rule["target"] is None:
|
|
93
|
+
return None
|
|
94
|
+
return rule
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _metric_to_spec(m, losses: list[Loss], chart: str) -> str | None:
|
|
98
|
+
if isinstance(m, str):
|
|
99
|
+
return m
|
|
100
|
+
if isinstance(m, dict):
|
|
101
|
+
if m.get("expressionType") == "SIMPLE":
|
|
102
|
+
agg = m.get("aggregate")
|
|
103
|
+
col = (m.get("column") or {}).get("column_name")
|
|
104
|
+
if agg in ADHOC_AGGREGATES and col:
|
|
105
|
+
if m.get("hasCustomLabel") and m.get("label"):
|
|
106
|
+
return f"{agg}({col}) AS {m['label']}"
|
|
107
|
+
return f"{agg}({col})"
|
|
108
|
+
if m.get("expressionType") == "SQL":
|
|
109
|
+
sql = (m.get("sqlExpression") or "").strip()
|
|
110
|
+
match = re.fullmatch(r"(SUM|AVG|COUNT|COUNT_DISTINCT|MIN|MAX)\(\s*(\*|\w+)\s*\)", sql, re.I)
|
|
111
|
+
if match:
|
|
112
|
+
base = f"{match.group(1).upper()}({match.group(2)})"
|
|
113
|
+
if m.get("hasCustomLabel") and m.get("label"):
|
|
114
|
+
return f"{base} AS {m['label']}"
|
|
115
|
+
return base
|
|
116
|
+
losses.append(Loss(chart, f"metric not representable, dropped: {m}"))
|
|
117
|
+
return None
|
|
118
|
+
losses.append(Loss(chart, f"unrecognized metric shape, dropped: {m!r}"))
|
|
119
|
+
return None
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _filters_to_spec(params: dict, losses: list[Loss], chart: str) -> list[dict]:
|
|
123
|
+
out = []
|
|
124
|
+
for f in params.get("adhoc_filters") or []:
|
|
125
|
+
if not isinstance(f, dict):
|
|
126
|
+
continue
|
|
127
|
+
if f.get("operator") == "TEMPORAL_RANGE":
|
|
128
|
+
continue # the time-range mechanism, not a data filter
|
|
129
|
+
if f.get("expressionType") == "SIMPLE" and f.get("clause", "WHERE") == "WHERE":
|
|
130
|
+
op = f.get("operator")
|
|
131
|
+
if op in _FILTER_OPS:
|
|
132
|
+
out.append({"column": f.get("subject"), "op": op, "comparator": f.get("comparator")})
|
|
133
|
+
continue
|
|
134
|
+
losses.append(Loss(chart, f"filter not representable, dropped: {f}"))
|
|
135
|
+
return [
|
|
136
|
+
{"column": f["column"], "op": f["op"], **({} if f["op"] in ("IS NULL", "IS NOT NULL") else {"value": f["comparator"]})}
|
|
137
|
+
for f in out
|
|
138
|
+
]
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _groupby_one(p: dict, losses: list[Loss], chart: str) -> str | None:
|
|
142
|
+
gb = p.get("groupby") or []
|
|
143
|
+
if isinstance(gb, str):
|
|
144
|
+
return gb
|
|
145
|
+
if len(gb) > 1:
|
|
146
|
+
losses.append(Loss(chart, f"multiple groupby {gb}; kept first only"))
|
|
147
|
+
return gb[0] if gb and isinstance(gb[0], str) else None
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _chart_to_spec(chart_yaml: dict, lookup: DatasetLookup, losses: list[Loss]) -> dict | None:
|
|
151
|
+
name = chart_yaml.get("slice_name") or "Unnamed"
|
|
152
|
+
viz = chart_yaml.get("viz_type")
|
|
153
|
+
p = chart_yaml.get("params") or {}
|
|
154
|
+
spec_type = REVERSE_VIZ.get(viz)
|
|
155
|
+
if viz == "echarts_timeseries_bar" and p.get(SDC_BAR_MARKER):
|
|
156
|
+
spec_type = "bar"
|
|
157
|
+
if viz == "histogram": # pre-6.x name
|
|
158
|
+
spec_type = "histogram"
|
|
159
|
+
if spec_type is None:
|
|
160
|
+
losses.append(Loss(name, f"viz_type {viz!r} outside spec surface; chart skipped"))
|
|
161
|
+
return None
|
|
162
|
+
ds = lookup(str(chart_yaml.get("dataset_uuid")))
|
|
163
|
+
if ds is None:
|
|
164
|
+
losses.append(Loss(name, f"dataset uuid {chart_yaml.get('dataset_uuid')} not resolvable; chart skipped"))
|
|
165
|
+
return None
|
|
166
|
+
out: dict = {"name": name, "type": spec_type, "dataset": ds}
|
|
167
|
+
flt = _filters_to_spec(p, losses, name)
|
|
168
|
+
if flt:
|
|
169
|
+
out["filters"] = flt
|
|
170
|
+
|
|
171
|
+
def metric_one(value) -> str | None:
|
|
172
|
+
return _metric_to_spec(value, losses, name)
|
|
173
|
+
|
|
174
|
+
def keep_row_limit() -> None:
|
|
175
|
+
if p.get("row_limit"):
|
|
176
|
+
out["row_limit"] = p["row_limit"]
|
|
177
|
+
|
|
178
|
+
if spec_type == "big_number_total":
|
|
179
|
+
m = metric_one(p.get("metric"))
|
|
180
|
+
if m is None:
|
|
181
|
+
return None
|
|
182
|
+
out["metric"] = m
|
|
183
|
+
if p.get("subheader"):
|
|
184
|
+
out["subtitle"] = p["subheader"]
|
|
185
|
+
if p.get("y_axis_format"):
|
|
186
|
+
out["number_format"] = p["y_axis_format"]
|
|
187
|
+
elif spec_type == "big_number_trend":
|
|
188
|
+
m = metric_one(p.get("metric"))
|
|
189
|
+
x = p.get("x_axis")
|
|
190
|
+
if m is None or not x:
|
|
191
|
+
losses.append(Loss(name, "big_number trend needs metric + x_axis; chart skipped"))
|
|
192
|
+
return None
|
|
193
|
+
out["metric"] = m
|
|
194
|
+
out["time_column"] = x
|
|
195
|
+
if p.get("time_grain_sqla"):
|
|
196
|
+
out["time_grain"] = p["time_grain_sqla"]
|
|
197
|
+
if p.get("y_axis_format"):
|
|
198
|
+
out["number_format"] = p["y_axis_format"]
|
|
199
|
+
elif spec_type in ("timeseries_line", "timeseries_bar", "timeseries_area", "timeseries_scatter"):
|
|
200
|
+
ms = [metric_one(m) for m in (p.get("metrics") or [])]
|
|
201
|
+
ms = [m for m in ms if m]
|
|
202
|
+
if not ms:
|
|
203
|
+
losses.append(Loss(name, "no representable metrics; chart skipped"))
|
|
204
|
+
return None
|
|
205
|
+
out["metrics"] = ms
|
|
206
|
+
x = p.get("x_axis")
|
|
207
|
+
if isinstance(x, dict):
|
|
208
|
+
x = x.get("sqlExpression") or x.get("label")
|
|
209
|
+
if not x:
|
|
210
|
+
losses.append(Loss(name, "no x_axis; chart skipped"))
|
|
211
|
+
return None
|
|
212
|
+
out["time_column"] = x
|
|
213
|
+
if p.get("time_grain_sqla"):
|
|
214
|
+
out["time_grain"] = p["time_grain_sqla"]
|
|
215
|
+
if p.get("time_range") and p["time_range"] != "No filter":
|
|
216
|
+
out["time_range"] = p["time_range"]
|
|
217
|
+
gb = _groupby_one(p, losses, name)
|
|
218
|
+
if gb:
|
|
219
|
+
out["groupby"] = gb
|
|
220
|
+
keep_row_limit()
|
|
221
|
+
elif spec_type == "bar":
|
|
222
|
+
ms = [m for m in (metric_one(m) for m in (p.get("metrics") or [])) if m]
|
|
223
|
+
x = p.get("x_axis")
|
|
224
|
+
if not ms or not x:
|
|
225
|
+
losses.append(Loss(name, "bar needs metrics + x_axis; chart skipped"))
|
|
226
|
+
return None
|
|
227
|
+
out["metrics"] = ms
|
|
228
|
+
out["x_column"] = x
|
|
229
|
+
if p.get("orientation") == "horizontal":
|
|
230
|
+
out["orientation"] = "horizontal"
|
|
231
|
+
gb = _groupby_one(p, losses, name)
|
|
232
|
+
if gb:
|
|
233
|
+
out["groupby"] = gb
|
|
234
|
+
keep_row_limit()
|
|
235
|
+
elif spec_type == "pie":
|
|
236
|
+
m = metric_one(p.get("metric"))
|
|
237
|
+
gb = _groupby_one(p, losses, name)
|
|
238
|
+
if m is None or not gb:
|
|
239
|
+
losses.append(Loss(name, "pie needs metric+groupby; chart skipped"))
|
|
240
|
+
return None
|
|
241
|
+
out["metric"] = m
|
|
242
|
+
out["groupby"] = gb
|
|
243
|
+
if p.get("donut"):
|
|
244
|
+
out["donut"] = True
|
|
245
|
+
keep_row_limit()
|
|
246
|
+
elif spec_type == "table":
|
|
247
|
+
if p.get("query_mode") == "raw" or p.get("all_columns"):
|
|
248
|
+
out["columns"] = p.get("all_columns") or []
|
|
249
|
+
if not out["columns"]:
|
|
250
|
+
losses.append(Loss(name, "raw table with no columns; chart skipped"))
|
|
251
|
+
return None
|
|
252
|
+
else:
|
|
253
|
+
ms = [m for m in (metric_one(m) for m in (p.get("metrics") or [])) if m]
|
|
254
|
+
out["metrics"] = ms or None
|
|
255
|
+
out["groupby"] = [g for g in (p.get("groupby") or []) if isinstance(g, str)] or None
|
|
256
|
+
if not out["metrics"] and not out["groupby"]:
|
|
257
|
+
losses.append(Loss(name, "aggregate table with no metrics/groupby; chart skipped"))
|
|
258
|
+
return None
|
|
259
|
+
keep_row_limit()
|
|
260
|
+
elif spec_type == "pivot_table":
|
|
261
|
+
ms = [m for m in (metric_one(m) for m in (p.get("metrics") or [])) if m]
|
|
262
|
+
rows = [c for c in (p.get("groupbyRows") or []) if isinstance(c, str)]
|
|
263
|
+
cols = [c for c in (p.get("groupbyColumns") or []) if isinstance(c, str)]
|
|
264
|
+
if not ms or not (rows or cols):
|
|
265
|
+
losses.append(Loss(name, "pivot needs metrics + rows/columns; chart skipped"))
|
|
266
|
+
return None
|
|
267
|
+
out["metrics"] = ms
|
|
268
|
+
if rows:
|
|
269
|
+
out["rows"] = rows
|
|
270
|
+
if cols:
|
|
271
|
+
out["columns"] = cols
|
|
272
|
+
if (p.get("aggregateFunction") or "Sum") != "Sum":
|
|
273
|
+
losses.append(Loss(name, f"pivot aggregateFunction {p['aggregateFunction']!r} not preserved (Sum on re-apply)"))
|
|
274
|
+
if p.get("combineMetric"):
|
|
275
|
+
out["combine_metric"] = True
|
|
276
|
+
if p.get("date_format"):
|
|
277
|
+
out["date_format"] = p["date_format"]
|
|
278
|
+
rules = []
|
|
279
|
+
for cf in p.get("conditional_formatting") or []:
|
|
280
|
+
rule = _format_to_spec(cf if isinstance(cf, dict) else {})
|
|
281
|
+
if rule is None:
|
|
282
|
+
losses.append(Loss(name, f"conditional format not representable, dropped: {cf}"))
|
|
283
|
+
else:
|
|
284
|
+
rules.append(rule)
|
|
285
|
+
if rules:
|
|
286
|
+
out["conditional_formatting"] = rules
|
|
287
|
+
keep_row_limit()
|
|
288
|
+
elif spec_type == "heatmap":
|
|
289
|
+
m = metric_one(p.get("metric"))
|
|
290
|
+
x = p.get("x_axis")
|
|
291
|
+
y = p.get("groupby") if isinstance(p.get("groupby"), str) else None
|
|
292
|
+
if m is None or not x or not y:
|
|
293
|
+
losses.append(Loss(name, "heatmap needs metric + x_axis + groupby; chart skipped"))
|
|
294
|
+
return None
|
|
295
|
+
out["metric"] = m
|
|
296
|
+
out["x_column"] = x
|
|
297
|
+
out["y_column"] = y
|
|
298
|
+
keep_row_limit()
|
|
299
|
+
elif spec_type == "histogram":
|
|
300
|
+
col = p.get("column")
|
|
301
|
+
if not col:
|
|
302
|
+
losses.append(Loss(name, "histogram without column; chart skipped"))
|
|
303
|
+
return None
|
|
304
|
+
out["column"] = col
|
|
305
|
+
if p.get("bins"):
|
|
306
|
+
out["bins"] = p["bins"]
|
|
307
|
+
gb = _groupby_one(p, losses, name)
|
|
308
|
+
if gb:
|
|
309
|
+
out["groupby"] = gb
|
|
310
|
+
keep_row_limit()
|
|
311
|
+
elif spec_type == "funnel":
|
|
312
|
+
m = metric_one(p.get("metric"))
|
|
313
|
+
gb = _groupby_one(p, losses, name)
|
|
314
|
+
if m is None or not gb:
|
|
315
|
+
losses.append(Loss(name, "funnel needs metric + groupby; chart skipped"))
|
|
316
|
+
return None
|
|
317
|
+
out["metric"] = m
|
|
318
|
+
out["groupby"] = gb
|
|
319
|
+
keep_row_limit()
|
|
320
|
+
elif spec_type == "treemap":
|
|
321
|
+
m = metric_one(p.get("metric"))
|
|
322
|
+
gb = [g for g in (p.get("groupby") or []) if isinstance(g, str)]
|
|
323
|
+
if m is None or not gb:
|
|
324
|
+
losses.append(Loss(name, "treemap needs metric + groupby; chart skipped"))
|
|
325
|
+
return None
|
|
326
|
+
out["metric"] = m
|
|
327
|
+
out["groupby"] = gb
|
|
328
|
+
keep_row_limit()
|
|
329
|
+
|
|
330
|
+
mapped_here = {"combineMetric", "conditional_formatting"} if spec_type == "pivot_table" else set()
|
|
331
|
+
unmapped = sorted(k for k in p if k not in _IGNORABLE and k not in mapped_here)
|
|
332
|
+
if unmapped:
|
|
333
|
+
losses.append(Loss(name, f"params not preserved: {unmapped}"))
|
|
334
|
+
return out
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _native_filters_to_spec(
|
|
338
|
+
metadata: dict, lookup: DatasetLookup, losses: list[Loss],
|
|
339
|
+
filter_uuids: dict[str, str] | None = None,
|
|
340
|
+
) -> list[dict]:
|
|
341
|
+
out = []
|
|
342
|
+
for nf in metadata.get("native_filter_configuration") or []:
|
|
343
|
+
name = nf.get("name") or nf.get("id") or "filter"
|
|
344
|
+
ftype = nf.get("filterType")
|
|
345
|
+
if ftype == "filter_select":
|
|
346
|
+
targets = nf.get("targets") or []
|
|
347
|
+
col = ((targets[0].get("column") or {}).get("name")) if targets else None
|
|
348
|
+
ds_uuid = targets[0].get("datasetUuid") if targets else None
|
|
349
|
+
ds = lookup(str(ds_uuid)) if ds_uuid else None
|
|
350
|
+
if not col or ds is None:
|
|
351
|
+
losses.append(Loss(f"filter:{name}", "select filter target not resolvable; dropped"))
|
|
352
|
+
continue
|
|
353
|
+
if filter_uuids is not None:
|
|
354
|
+
filter_uuids[f"filter:{name}"] = str(ds_uuid)
|
|
355
|
+
f: dict = {"type": "select", "name": name, "dataset": ds, "column": col}
|
|
356
|
+
multi = (nf.get("controlValues") or {}).get("multiSelect", True)
|
|
357
|
+
if multi is False:
|
|
358
|
+
f["multi"] = False
|
|
359
|
+
out.append(f)
|
|
360
|
+
elif ftype == "filter_range":
|
|
361
|
+
targets = nf.get("targets") or []
|
|
362
|
+
col = ((targets[0].get("column") or {}).get("name")) if targets else None
|
|
363
|
+
ds_uuid = targets[0].get("datasetUuid") if targets else None
|
|
364
|
+
ds = lookup(str(ds_uuid)) if ds_uuid else None
|
|
365
|
+
if not col or ds is None:
|
|
366
|
+
losses.append(Loss(f"filter:{name}", "range filter target not resolvable; dropped"))
|
|
367
|
+
continue
|
|
368
|
+
if filter_uuids is not None:
|
|
369
|
+
filter_uuids[f"filter:{name}"] = str(ds_uuid)
|
|
370
|
+
f = {"type": "range", "name": name, "dataset": ds, "column": col}
|
|
371
|
+
value = ((nf.get("defaultDataMask") or {}).get("filterState") or {}).get("value")
|
|
372
|
+
if isinstance(value, list) and len(value) == 2:
|
|
373
|
+
if value[0] is not None:
|
|
374
|
+
f["ge"] = value[0]
|
|
375
|
+
if value[1] is not None:
|
|
376
|
+
f["le"] = value[1]
|
|
377
|
+
scoped = nf.get("sdc_scope_charts")
|
|
378
|
+
if scoped:
|
|
379
|
+
# Tool-born filters carry their name-based scope; the numeric
|
|
380
|
+
# live scope is derived from it by apply's scope stage.
|
|
381
|
+
f["charts"] = list(scoped)
|
|
382
|
+
elif (nf.get("scope") or {}).get("excluded"):
|
|
383
|
+
losses.append(Loss(
|
|
384
|
+
f"filter:{name}",
|
|
385
|
+
"chart scope not preserved (live scopes are numeric slice ids; "
|
|
386
|
+
"re-declare `charts` by name in the spec)",
|
|
387
|
+
))
|
|
388
|
+
out.append(f)
|
|
389
|
+
continue # range preserves its default; skip the default-loss check
|
|
390
|
+
elif ftype == "filter_time":
|
|
391
|
+
f = {"type": "time_range", "name": name}
|
|
392
|
+
value = ((nf.get("defaultDataMask") or {}).get("filterState") or {}).get("value")
|
|
393
|
+
if isinstance(value, str) and value:
|
|
394
|
+
f["default"] = value
|
|
395
|
+
out.append(f)
|
|
396
|
+
continue # default preserved; skip the default-loss check
|
|
397
|
+
out.append(f)
|
|
398
|
+
else:
|
|
399
|
+
losses.append(Loss(f"filter:{name}", f"filterType {ftype!r} outside spec surface; dropped"))
|
|
400
|
+
continue
|
|
401
|
+
dm = nf.get("defaultDataMask") or {}
|
|
402
|
+
if dm.get("filterState") or dm.get("extraFormData"):
|
|
403
|
+
losses.append(Loss(f"filter:{name}", "default value not preserved"))
|
|
404
|
+
return out
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _walk_rows(position: dict, children: list[str], kept_names: set[str],
|
|
408
|
+
losses: list[Loss], geometry: dict[str, dict]) -> list[list]:
|
|
409
|
+
"""Convert ROW children into spec rows (chart names + markdown blocks)."""
|
|
410
|
+
rows: list[list] = []
|
|
411
|
+
|
|
412
|
+
def handle(children_ids: list[str], depth: int = 0) -> None:
|
|
413
|
+
if depth > 10:
|
|
414
|
+
return
|
|
415
|
+
for cid in children_ids:
|
|
416
|
+
node = position.get(cid)
|
|
417
|
+
if not node:
|
|
418
|
+
continue
|
|
419
|
+
t = node.get("type")
|
|
420
|
+
if t == "ROW":
|
|
421
|
+
row: list = []
|
|
422
|
+
for ch_id in node.get("children", []):
|
|
423
|
+
ch = position.get(ch_id) or {}
|
|
424
|
+
meta = ch.get("meta") or {}
|
|
425
|
+
if ch.get("type") == "CHART":
|
|
426
|
+
nm = meta.get("sliceName")
|
|
427
|
+
if nm and nm in kept_names:
|
|
428
|
+
row.append(nm)
|
|
429
|
+
geometry[nm] = {"width": meta.get("width"), "height": meta.get("height")}
|
|
430
|
+
elif nm:
|
|
431
|
+
losses.append(Loss("layout", f"chart {nm!r} in layout but not decompilable; removed from row"))
|
|
432
|
+
elif ch.get("type") == "MARKDOWN":
|
|
433
|
+
block: dict = {"markdown": meta.get("code") or ""}
|
|
434
|
+
if meta.get("width"):
|
|
435
|
+
block["width"] = max(1, min(12, int(meta["width"])))
|
|
436
|
+
if meta.get("height"):
|
|
437
|
+
block["height"] = max(1, round(int(meta["height"]) / ROW_UNITS_PER_SPEC_UNIT))
|
|
438
|
+
if block["markdown"]:
|
|
439
|
+
row.append(block)
|
|
440
|
+
else:
|
|
441
|
+
losses.append(Loss("layout", "empty MARKDOWN dropped"))
|
|
442
|
+
elif ch.get("type") == "COLUMN":
|
|
443
|
+
losses.append(Loss("layout", "COLUMN (vertical stacking) flattened: children pulled up"))
|
|
444
|
+
handle(ch.get("children", []), depth + 1)
|
|
445
|
+
else:
|
|
446
|
+
losses.append(Loss("layout", f"{ch.get('type')} element dropped from a row"))
|
|
447
|
+
if row:
|
|
448
|
+
rows.append(row)
|
|
449
|
+
elif t == "CHART":
|
|
450
|
+
meta = node.get("meta") or {}
|
|
451
|
+
nm = meta.get("sliceName")
|
|
452
|
+
if nm and nm in kept_names:
|
|
453
|
+
rows.append([nm])
|
|
454
|
+
geometry[nm] = {"width": meta.get("width"), "height": meta.get("height")}
|
|
455
|
+
elif t in ("TABS", "TAB"):
|
|
456
|
+
# mixed/nested tabs at this level are handled by the caller;
|
|
457
|
+
# reaching here means nested tabs inside a tab -> flatten
|
|
458
|
+
losses.append(Loss("layout", f"nested {t} flattened"))
|
|
459
|
+
handle(node.get("children", []), depth + 1)
|
|
460
|
+
else:
|
|
461
|
+
losses.append(Loss("layout", f"{t or cid} element dropped"))
|
|
462
|
+
|
|
463
|
+
handle(children)
|
|
464
|
+
return rows
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def decompile_bundle(zip_bytes: bytes, lookup: DatasetLookup) -> DecompileResult:
|
|
468
|
+
losses: list[Loss] = []
|
|
469
|
+
zf = zipfile.ZipFile(io.BytesIO(zip_bytes))
|
|
470
|
+
dash_files = [n for n in zf.namelist() if "/dashboards/" in n and n.endswith(".yaml")]
|
|
471
|
+
if not dash_files:
|
|
472
|
+
raise ValueError("no dashboards/*.yaml in bundle")
|
|
473
|
+
if len(dash_files) > 1:
|
|
474
|
+
losses.append(Loss("dashboard", f"bundle has {len(dash_files)} dashboards; decompiling the first only"))
|
|
475
|
+
dash = yaml.safe_load(zf.read(dash_files[0]))
|
|
476
|
+
|
|
477
|
+
charts_by_name: dict[str, dict] = {}
|
|
478
|
+
dataset_uuids: dict[str, str] = {}
|
|
479
|
+
for n in zf.namelist():
|
|
480
|
+
if "/charts/" in n and n.endswith(".yaml"):
|
|
481
|
+
cy = yaml.safe_load(zf.read(n))
|
|
482
|
+
spec_chart = _chart_to_spec(cy, lookup, losses)
|
|
483
|
+
if spec_chart:
|
|
484
|
+
if spec_chart["name"] in charts_by_name:
|
|
485
|
+
losses.append(Loss(spec_chart["name"], "duplicate slice_name in bundle; suffixed to keep uuid seeds unique"))
|
|
486
|
+
spec_chart["name"] = f"{spec_chart['name']} (2)"
|
|
487
|
+
charts_by_name[spec_chart["name"]] = spec_chart
|
|
488
|
+
dataset_uuids[spec_chart["name"]] = str(cy.get("dataset_uuid"))
|
|
489
|
+
|
|
490
|
+
title = dash.get("dashboard_title") or "Untitled"
|
|
491
|
+
slug = dash.get("slug")
|
|
492
|
+
if not slug:
|
|
493
|
+
slug = re.sub(r"-+", "-", re.sub(r"[^a-z0-9]", "-", title.lower())).strip("-") or "untitled"
|
|
494
|
+
losses.append(Loss("dashboard", f"no slug on source dashboard; derived {slug!r} (re-apply will NOT overwrite the original)"))
|
|
495
|
+
|
|
496
|
+
position = dash.get("position") or {}
|
|
497
|
+
geometry: dict[str, dict] = {}
|
|
498
|
+
kept = set(charts_by_name)
|
|
499
|
+
grid = position.get("GRID_ID") or {}
|
|
500
|
+
grid_children = grid.get("children", [])
|
|
501
|
+
top_types = {(position.get(c) or {}).get("type") for c in grid_children}
|
|
502
|
+
|
|
503
|
+
layout: dict
|
|
504
|
+
if top_types and top_types <= {"TABS"}:
|
|
505
|
+
tabs = []
|
|
506
|
+
for tabs_id in grid_children:
|
|
507
|
+
for tab_id in (position.get(tabs_id) or {}).get("children", []):
|
|
508
|
+
tab_node = position.get(tab_id) or {}
|
|
509
|
+
tab_rows = _walk_rows(position, tab_node.get("children", []), kept, losses, geometry)
|
|
510
|
+
if tab_rows:
|
|
511
|
+
tabs.append({"title": (tab_node.get("meta") or {}).get("text") or "Tab", "rows": tab_rows})
|
|
512
|
+
else:
|
|
513
|
+
losses.append(Loss("layout", f"tab {(tab_node.get('meta') or {}).get('text')!r} had no representable content; dropped"))
|
|
514
|
+
layout = {"tabs": tabs} if tabs else {"rows": []}
|
|
515
|
+
else:
|
|
516
|
+
if "TABS" in top_types:
|
|
517
|
+
losses.append(Loss("layout", "mixed rows + tabs at top level; tabs flattened into rows"))
|
|
518
|
+
rows = _walk_rows(position, grid_children, kept, losses, geometry)
|
|
519
|
+
layout = {"rows": rows}
|
|
520
|
+
|
|
521
|
+
all_rows = layout.get("rows") if "rows" in layout else [r for t in layout["tabs"] for r in t["rows"]]
|
|
522
|
+
placed = {x for row in (all_rows or []) for x in row if isinstance(x, str)}
|
|
523
|
+
unplaced_target = layout.get("rows") if "rows" in layout else (layout["tabs"][0]["rows"] if layout.get("tabs") else None)
|
|
524
|
+
for name in sorted(kept - placed):
|
|
525
|
+
losses.append(Loss("layout", f"chart {name!r} not found in layout; appended as its own row"))
|
|
526
|
+
if unplaced_target is None:
|
|
527
|
+
layout = {"rows": [[name]]}
|
|
528
|
+
unplaced_target = layout["rows"]
|
|
529
|
+
else:
|
|
530
|
+
unplaced_target.append([name])
|
|
531
|
+
|
|
532
|
+
for name, geo in geometry.items():
|
|
533
|
+
c = charts_by_name.get(name)
|
|
534
|
+
if not c:
|
|
535
|
+
continue
|
|
536
|
+
if geo.get("width"):
|
|
537
|
+
c["width"] = max(1, min(12, int(geo["width"])))
|
|
538
|
+
if geo.get("height"):
|
|
539
|
+
h = max(1, round(int(geo["height"]) / ROW_UNITS_PER_SPEC_UNIT))
|
|
540
|
+
c["height"] = h
|
|
541
|
+
if int(geo["height"]) != h * ROW_UNITS_PER_SPEC_UNIT:
|
|
542
|
+
losses.append(Loss(name, f"height {geo['height']} rounded to {h * ROW_UNITS_PER_SPEC_UNIT} row units"))
|
|
543
|
+
|
|
544
|
+
# Row overflow guard: source rows can exceed 12 units after flattening.
|
|
545
|
+
def width_of(item) -> int:
|
|
546
|
+
if isinstance(item, str):
|
|
547
|
+
return charts_by_name.get(item, {}).get("width") or 0
|
|
548
|
+
return item.get("width") or 0
|
|
549
|
+
|
|
550
|
+
all_rows = layout.get("rows") if "rows" in layout else [r for t in layout["tabs"] for r in t["rows"]]
|
|
551
|
+
for i, row in enumerate(all_rows or []):
|
|
552
|
+
total = sum(width_of(x) for x in row)
|
|
553
|
+
if total > 12:
|
|
554
|
+
losses.append(Loss("layout", f"row {i} widths sum to {total} > 12; widths cleared, will auto-split"))
|
|
555
|
+
for x in row:
|
|
556
|
+
if isinstance(x, str):
|
|
557
|
+
charts_by_name.get(x, {}).pop("width", None)
|
|
558
|
+
else:
|
|
559
|
+
x.pop("width", None)
|
|
560
|
+
|
|
561
|
+
ordered_names: list[str] = []
|
|
562
|
+
for row in (all_rows or []):
|
|
563
|
+
for x in row:
|
|
564
|
+
if isinstance(x, str) and x in charts_by_name:
|
|
565
|
+
ordered_names.append(x)
|
|
566
|
+
ordered = [charts_by_name[n] for n in ordered_names]
|
|
567
|
+
|
|
568
|
+
spec = {
|
|
569
|
+
"spec_version": "1",
|
|
570
|
+
"dashboard": {"title": title, "slug": slug},
|
|
571
|
+
"charts": ordered,
|
|
572
|
+
"layout": layout,
|
|
573
|
+
}
|
|
574
|
+
# Filter dataset identities ride along under "filter:<name>" keys so plan
|
|
575
|
+
# can compare filters by resolved uuid, exactly as it does for charts.
|
|
576
|
+
filters = _native_filters_to_spec(dash.get("metadata") or {}, lookup, losses,
|
|
577
|
+
filter_uuids=dataset_uuids)
|
|
578
|
+
if filters:
|
|
579
|
+
spec["filters"] = filters
|
|
580
|
+
if not ordered:
|
|
581
|
+
losses.append(Loss("dashboard", "no representable charts; spec is not valid for apply"))
|
|
582
|
+
return DecompileResult(spec=spec, losses=losses, dataset_uuids=dataset_uuids)
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def live_dataset_lookup(client) -> DatasetLookup:
|
|
586
|
+
"""uuid -> triple, resolved lazily against the live instance."""
|
|
587
|
+
cache: dict[str, dict] | None = None
|
|
588
|
+
|
|
589
|
+
def lookup(u: str) -> dict | None:
|
|
590
|
+
nonlocal cache
|
|
591
|
+
if cache is None:
|
|
592
|
+
cache = {}
|
|
593
|
+
page = 0
|
|
594
|
+
while True:
|
|
595
|
+
out = client.get(
|
|
596
|
+
"/api/v1/dataset/",
|
|
597
|
+
q={"columns": ["uuid", "table_name", "schema", "database.database_name"],
|
|
598
|
+
"page": page, "page_size": 100},
|
|
599
|
+
)["result"]
|
|
600
|
+
if not out:
|
|
601
|
+
break
|
|
602
|
+
for d in out:
|
|
603
|
+
cache[str(d["uuid"])] = {
|
|
604
|
+
"database": (d.get("database") or {}).get("database_name"),
|
|
605
|
+
"schema": d.get("schema") or None,
|
|
606
|
+
"table": d["table_name"],
|
|
607
|
+
}
|
|
608
|
+
page += 1
|
|
609
|
+
if page > 200:
|
|
610
|
+
break
|
|
611
|
+
return cache.get(u)
|
|
612
|
+
|
|
613
|
+
return lookup
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def decompile_live(slug_or_id: str, client) -> DecompileResult:
|
|
617
|
+
if slug_or_id.isdigit():
|
|
618
|
+
did = int(slug_or_id)
|
|
619
|
+
else:
|
|
620
|
+
dash = client.find_dashboard_by_slug(slug_or_id)
|
|
621
|
+
if dash is None:
|
|
622
|
+
raise ValueError(f"no dashboard with slug {slug_or_id!r}")
|
|
623
|
+
did = dash["id"]
|
|
624
|
+
blob = client.export_dashboard(did)
|
|
625
|
+
return decompile_bundle(blob, live_dataset_lookup(client))
|
chartwright/ids.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Deterministic identity. uuid5 under a fixed project namespace, seeded from
|
|
2
|
+
spec paths, so recompiles are byte-stable and re-apply updates in place.
|
|
3
|
+
|
|
4
|
+
Renaming a chart mints a new UUID; apply deletes owned charts that leave the
|
|
5
|
+
spec, so renames propagate without leaving orphans.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import uuid
|
|
9
|
+
|
|
10
|
+
# The identity seed is FROZEN at the project's original name: every dashboard
|
|
11
|
+
# and chart uuid ever created derives from it, so renaming it would make the
|
|
12
|
+
# tool refuse to touch its own dashboards. The product renamed; this cannot.
|
|
13
|
+
NAMESPACE = uuid.uuid5(uuid.NAMESPACE_URL, "superset-dashboard-compiler")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def dashboard_uuid(slug: str) -> uuid.UUID:
|
|
17
|
+
return uuid.uuid5(NAMESPACE, slug)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def chart_uuid(slug: str, chart_name: str) -> uuid.UUID:
|
|
21
|
+
return uuid.uuid5(NAMESPACE, f"{slug}/chart/{chart_name}")
|