perspective-python 4.5.2__cp311-abi3-pyemscripten_2025_0_wasm32.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- perspective/__init__.py +403 -0
- perspective/extension/finos-perspective-nbextension.json +5 -0
- perspective/handlers/__init__.py +11 -0
- perspective/handlers/aiohttp.py +69 -0
- perspective/handlers/starlette.py +63 -0
- perspective/handlers/tornado.py +192 -0
- perspective/perspective.abi3.so +0 -0
- perspective/templates/exported_widget.html.template +35 -0
- perspective/tests/__init__.py +11 -0
- perspective/tests/async/test_async_client.py +83 -0
- perspective/tests/async/test_websocket_client.py +124 -0
- perspective/tests/conftest.py +263 -0
- perspective/tests/core/__init__.py +11 -0
- perspective/tests/core/test_async.py +351 -0
- perspective/tests/multi_threaded/__init__.py +11 -0
- perspective/tests/multi_threaded/test_multi_threaded.py +201 -0
- perspective/tests/server/__init__.py +11 -0
- perspective/tests/server/test_server.py +1016 -0
- perspective/tests/server/test_session.py +110 -0
- perspective/tests/table/__init__.py +11 -0
- perspective/tests/table/arrow/date32.arrow +0 -0
- perspective/tests/table/arrow/date64.arrow +0 -0
- perspective/tests/table/arrow/dict.arrow +0 -0
- perspective/tests/table/arrow/dict_update.arrow +0 -0
- perspective/tests/table/arrow/int_float_str.arrow +0 -0
- perspective/tests/table/arrow/int_float_str_file.arrow +0 -0
- perspective/tests/table/arrow/int_float_str_update.arrow +0 -0
- perspective/tests/table/object_sequence.py +402 -0
- perspective/tests/table/test_column_paths.py +89 -0
- perspective/tests/table/test_delete.py +124 -0
- perspective/tests/table/test_exception.py +65 -0
- perspective/tests/table/test_join.py +115 -0
- perspective/tests/table/test_leaks.py +105 -0
- perspective/tests/table/test_ports.py +178 -0
- perspective/tests/table/test_remove.py +102 -0
- perspective/tests/table/test_table.py +641 -0
- perspective/tests/table/test_table_arrow.py +500 -0
- perspective/tests/table/test_table_datetime.py +2409 -0
- perspective/tests/table/test_table_infer.py +201 -0
- perspective/tests/table/test_table_limit.py +45 -0
- perspective/tests/table/test_table_numpy.py +1022 -0
- perspective/tests/table/test_table_page_to_disk.py +304 -0
- perspective/tests/table/test_table_pandas.py +1018 -0
- perspective/tests/table/test_table_polars.py +251 -0
- perspective/tests/table/test_table_view_table.py +130 -0
- perspective/tests/table/test_to_arrow.py +417 -0
- perspective/tests/table/test_to_arrow_lz4.py +32 -0
- perspective/tests/table/test_to_format.py +1024 -0
- perspective/tests/table/test_to_polars.py +26 -0
- perspective/tests/table/test_update.py +536 -0
- perspective/tests/table/test_update_arrow.py +980 -0
- perspective/tests/table/test_update_pandas.py +211 -0
- perspective/tests/table/test_view.py +2444 -0
- perspective/tests/table/test_view_expression.py +1940 -0
- perspective/tests/test_dependencies.py +53 -0
- perspective/tests/viewer/__init__.py +11 -0
- perspective/tests/viewer/test_viewer.py +246 -0
- perspective/tests/virtual_servers/__init__.py +12 -0
- perspective/tests/virtual_servers/test_coerce_types.py +209 -0
- perspective/tests/virtual_servers/test_duckdb.py +908 -0
- perspective/tests/virtual_servers/test_polars.py +1032 -0
- perspective/tests/widget/__init__.py +11 -0
- perspective/tests/widget/test_widget.py +278 -0
- perspective/tests/widget/test_widget_pandas.py +453 -0
- perspective/virtual_servers/__init__.py +142 -0
- perspective/virtual_servers/clickhouse.py +240 -0
- perspective/virtual_servers/duckdb.py +236 -0
- perspective/virtual_servers/polars.py +710 -0
- perspective/widget/__init__.py +349 -0
- perspective/widget/viewer/__init__.py +15 -0
- perspective/widget/viewer/validate.py +22 -0
- perspective/widget/viewer/viewer.py +343 -0
- perspective/widget/viewer/viewer_traitlets.py +101 -0
- perspective_python-4.5.2.dist-info/METADATA +29 -0
- perspective_python-4.5.2.dist-info/RECORD +79 -0
- perspective_python-4.5.2.dist-info/WHEEL +4 -0
- perspective_python-4.5.2.dist-info/licenses/LICENSE.md +193 -0
- perspective_python-4.5.2.dist-info/licenses/LICENSE_THIRDPARTY_cargo.yml +19255 -0
- perspective_python-4.5.2.dist-info/sboms/perspective-python.cyclonedx.json +5546 -0
|
@@ -0,0 +1,710 @@
|
|
|
1
|
+
# ┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┓
|
|
2
|
+
# ┃ ██████ ██████ ██████ █ █ █ █ █ █▄ ▀███ █ ┃
|
|
3
|
+
# ┃ ▄▄▄▄▄█ █▄▄▄▄▄ ▄▄▄▄▄█ ▀▀▀▀▀█▀▀▀▀▀ █ ▀▀▀▀▀█ ████████▌▐███ ███▄ ▀█ █ ▀▀▀▀▀ ┃
|
|
4
|
+
# ┃ █▀▀▀▀▀ █▀▀▀▀▀ █▀██▀▀ ▄▄▄▄▄ █ ▄▄▄▄▄█ ▄▄▄▄▄█ ████████▌▐███ █████▄ █ ▄▄▄▄▄ ┃
|
|
5
|
+
# ┃ █ ██████ █ ▀█▄ █ ██████ █ ███▌▐███ ███████▄ █ ┃
|
|
6
|
+
# ┣━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┫
|
|
7
|
+
# ┃ Copyright (c) 2017, the Perspective Authors. ┃
|
|
8
|
+
# ┃ ╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌╌ ┃
|
|
9
|
+
# ┃ This file is part of the Perspective library, distributed under the terms ┃
|
|
10
|
+
# ┃ of the [Apache License 2.0](https://www.apache.org/licenses/LICENSE-2.0). ┃
|
|
11
|
+
# ┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┛
|
|
12
|
+
|
|
13
|
+
import logging
|
|
14
|
+
import polars as pl
|
|
15
|
+
import perspective
|
|
16
|
+
|
|
17
|
+
from datetime import datetime
|
|
18
|
+
import re
|
|
19
|
+
|
|
20
|
+
from perspective.virtual_servers import VirtualServerHandler
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
NUMBER_AGGS = [
|
|
25
|
+
"sum",
|
|
26
|
+
"count",
|
|
27
|
+
"any_value",
|
|
28
|
+
"avg",
|
|
29
|
+
"mean",
|
|
30
|
+
"max",
|
|
31
|
+
"min",
|
|
32
|
+
"first",
|
|
33
|
+
"last",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
STRING_AGGS = [
|
|
37
|
+
"count",
|
|
38
|
+
"any_value",
|
|
39
|
+
"first",
|
|
40
|
+
"last",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
FILTER_OPS = [
|
|
44
|
+
"==",
|
|
45
|
+
"!=",
|
|
46
|
+
">=",
|
|
47
|
+
"<=",
|
|
48
|
+
">",
|
|
49
|
+
"<",
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
AGG_MAP = {
|
|
53
|
+
"sum": lambda e: e.sum(),
|
|
54
|
+
"count": lambda e: e.count(),
|
|
55
|
+
"avg": lambda e: e.mean(),
|
|
56
|
+
"mean": lambda e: e.mean(),
|
|
57
|
+
"min": lambda e: e.min(),
|
|
58
|
+
"max": lambda e: e.max(),
|
|
59
|
+
"first": lambda e: e.first(),
|
|
60
|
+
"last": lambda e: e.last(),
|
|
61
|
+
"any_value": lambda e: e.first(),
|
|
62
|
+
"arbitrary": lambda e: e.first(),
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class PolarsVirtualSession:
|
|
67
|
+
def __init__(self, callback, tables):
|
|
68
|
+
self.session = perspective.VirtualServer(PolarsVirtualServerHandler(tables))
|
|
69
|
+
self.callback = callback
|
|
70
|
+
|
|
71
|
+
def handle_request(self, msg):
|
|
72
|
+
self.callback(self.session.handle_request(msg))
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class PolarsVirtualServer:
|
|
76
|
+
def __init__(self, tables):
|
|
77
|
+
self.tables = tables
|
|
78
|
+
|
|
79
|
+
def new_session(self, callback):
|
|
80
|
+
return PolarsVirtualSession(callback, self.tables)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class PolarsVirtualServerHandler(VirtualServerHandler):
|
|
84
|
+
"""
|
|
85
|
+
An implementation of a `perspective.VirtualServerHandler` for Polars.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
def __init__(self, tables):
|
|
89
|
+
self.tables = tables
|
|
90
|
+
self.views = {}
|
|
91
|
+
self.view_schemas = {}
|
|
92
|
+
|
|
93
|
+
def get_features(self):
|
|
94
|
+
return {
|
|
95
|
+
"group_by": True,
|
|
96
|
+
"split_by": True,
|
|
97
|
+
"sort": True,
|
|
98
|
+
"expressions": True,
|
|
99
|
+
"group_rollup_mode": ["rollup", "flat", "total"],
|
|
100
|
+
"filter_ops": {
|
|
101
|
+
"integer": FILTER_OPS,
|
|
102
|
+
"float": FILTER_OPS,
|
|
103
|
+
"string": FILTER_OPS,
|
|
104
|
+
"boolean": ["==", "!="],
|
|
105
|
+
"date": FILTER_OPS,
|
|
106
|
+
"datetime": FILTER_OPS,
|
|
107
|
+
},
|
|
108
|
+
"aggregates": {
|
|
109
|
+
"integer": NUMBER_AGGS,
|
|
110
|
+
"float": NUMBER_AGGS,
|
|
111
|
+
"string": STRING_AGGS,
|
|
112
|
+
"boolean": STRING_AGGS,
|
|
113
|
+
"date": STRING_AGGS,
|
|
114
|
+
"datetime": STRING_AGGS,
|
|
115
|
+
},
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
def get_hosted_tables(self):
|
|
119
|
+
return list(self.tables.keys())
|
|
120
|
+
|
|
121
|
+
def table_schema(self, table_name, config=None):
|
|
122
|
+
df = self.tables[table_name]
|
|
123
|
+
schema = {}
|
|
124
|
+
for col_name, dtype in df.schema.items():
|
|
125
|
+
if not col_name.startswith("__"):
|
|
126
|
+
schema[col_name] = polars_type_to_psp(dtype)
|
|
127
|
+
return schema
|
|
128
|
+
|
|
129
|
+
def table_size(self, table_name):
|
|
130
|
+
return self.tables[table_name].height
|
|
131
|
+
|
|
132
|
+
def view_schema(self, view_name, config):
|
|
133
|
+
if view_name in self.view_schemas:
|
|
134
|
+
return self.view_schemas[view_name]
|
|
135
|
+
return self.table_schema(view_name)
|
|
136
|
+
|
|
137
|
+
def view_size(self, view_name):
|
|
138
|
+
if view_name in self.views:
|
|
139
|
+
return self.views[view_name].height
|
|
140
|
+
return self.table_size(view_name)
|
|
141
|
+
|
|
142
|
+
def table_validate_expression(self, table_name, expression):
|
|
143
|
+
df = self.tables.get(table_name)
|
|
144
|
+
if df is None:
|
|
145
|
+
return None
|
|
146
|
+
expr = parse_expression(expression)
|
|
147
|
+
result = df.select(expr.alias("__expr__"))
|
|
148
|
+
return polars_type_to_psp(result["__expr__"].dtype)
|
|
149
|
+
|
|
150
|
+
def table_make_view(self, table_name, view_name, config):
|
|
151
|
+
start = datetime.now()
|
|
152
|
+
df = self.tables[table_name]
|
|
153
|
+
group_by = config.get("group_by", [])
|
|
154
|
+
columns = [c for c in config.get("columns", []) if c is not None]
|
|
155
|
+
aggregates = config.get("aggregates", {})
|
|
156
|
+
sort = config.get("sort", [])
|
|
157
|
+
filters = config.get("filter", [])
|
|
158
|
+
split_by = config.get("split_by", [])
|
|
159
|
+
expressions = config.get("expressions", {})
|
|
160
|
+
group_rollup_mode = config.get("group_rollup_mode", "rollup")
|
|
161
|
+
|
|
162
|
+
if expressions:
|
|
163
|
+
for expr_name, expr_str in expressions.items():
|
|
164
|
+
expr = parse_expression(expr_str)
|
|
165
|
+
df = df.with_columns(expr.alias(expr_name))
|
|
166
|
+
|
|
167
|
+
df = apply_filters(df, filters)
|
|
168
|
+
|
|
169
|
+
col_alias = lambda c: c.replace("_", "-")
|
|
170
|
+
is_flat = group_rollup_mode == "flat"
|
|
171
|
+
is_total = group_rollup_mode == "total"
|
|
172
|
+
|
|
173
|
+
if is_total:
|
|
174
|
+
if split_by:
|
|
175
|
+
result = build_split_by_total(
|
|
176
|
+
df, split_by, columns, aggregates, col_alias
|
|
177
|
+
)
|
|
178
|
+
else:
|
|
179
|
+
result = build_total(df, columns, aggregates, col_alias)
|
|
180
|
+
elif split_by and group_by:
|
|
181
|
+
if is_flat:
|
|
182
|
+
result = build_split_by_grouped_flat(
|
|
183
|
+
df, group_by, split_by, columns, aggregates, col_alias
|
|
184
|
+
)
|
|
185
|
+
result = apply_sort_split_by_flat(
|
|
186
|
+
result, sort, columns, group_by, split_by
|
|
187
|
+
)
|
|
188
|
+
else:
|
|
189
|
+
result = build_split_by_grouped(
|
|
190
|
+
df, group_by, split_by, columns, aggregates, col_alias
|
|
191
|
+
)
|
|
192
|
+
result = apply_sort_grouped(result, sort, group_by, col_alias)
|
|
193
|
+
elif split_by:
|
|
194
|
+
result = build_split_by_flat(df, split_by, columns, col_alias)
|
|
195
|
+
result = apply_sort_flat(result, sort, col_alias)
|
|
196
|
+
elif group_by:
|
|
197
|
+
if is_flat:
|
|
198
|
+
result = build_flat_group_by(
|
|
199
|
+
df, group_by, columns, aggregates, col_alias
|
|
200
|
+
)
|
|
201
|
+
result = apply_sort_flat(result, sort, col_alias)
|
|
202
|
+
else:
|
|
203
|
+
result = build_rollup(df, group_by, columns, aggregates, col_alias)
|
|
204
|
+
result = apply_sort_grouped(result, sort, group_by, col_alias)
|
|
205
|
+
else:
|
|
206
|
+
select_exprs = [pl.col(c).alias(col_alias(c)) for c in columns]
|
|
207
|
+
result = df.select(select_exprs)
|
|
208
|
+
result = apply_sort_flat(result, sort, col_alias)
|
|
209
|
+
|
|
210
|
+
self.views[view_name] = result
|
|
211
|
+
self.view_schemas[view_name] = compute_view_schema(result)
|
|
212
|
+
logger.debug(
|
|
213
|
+
f"{datetime.now() - start} table_make_view {table_name} -> {view_name}"
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
def view_delete(self, view_name):
|
|
217
|
+
self.views.pop(view_name, None)
|
|
218
|
+
self.view_schemas.pop(view_name, None)
|
|
219
|
+
|
|
220
|
+
def view_get_min_max(self, view_name, column_name, config):
|
|
221
|
+
df = self.views[view_name]
|
|
222
|
+
col = df[column_name]
|
|
223
|
+
min_val = col.min()
|
|
224
|
+
max_val = col.max()
|
|
225
|
+
return (min_val, max_val)
|
|
226
|
+
|
|
227
|
+
def view_get_data(self, view_name, config, schema, viewport, data):
|
|
228
|
+
df = self.views.get(view_name)
|
|
229
|
+
if df is None:
|
|
230
|
+
return
|
|
231
|
+
|
|
232
|
+
group_by = config.get("group_by", [])
|
|
233
|
+
split_by = config.get("split_by", [])
|
|
234
|
+
group_rollup_mode = config.get("group_rollup_mode", "rollup")
|
|
235
|
+
is_split_by = len(split_by) > 0
|
|
236
|
+
is_flat = group_rollup_mode == "flat"
|
|
237
|
+
|
|
238
|
+
start_row = viewport.get("start_row", 0) or 0
|
|
239
|
+
end_row = viewport.get("end_row") or df.height
|
|
240
|
+
start_col = viewport.get("start_col", 0) or 0
|
|
241
|
+
end_col = viewport.get("end_col")
|
|
242
|
+
|
|
243
|
+
length = min(end_row, df.height) - start_row
|
|
244
|
+
if length <= 0:
|
|
245
|
+
return
|
|
246
|
+
df_slice = df.slice(start_row, length)
|
|
247
|
+
|
|
248
|
+
data_columns = [c for c in schema.keys() if not c.startswith("__")]
|
|
249
|
+
if end_col is not None:
|
|
250
|
+
data_columns = data_columns[start_col:end_col]
|
|
251
|
+
else:
|
|
252
|
+
data_columns = data_columns[start_col:]
|
|
253
|
+
|
|
254
|
+
has_group_by = len(group_by) > 0
|
|
255
|
+
has_grouping_id = has_group_by and not is_flat
|
|
256
|
+
|
|
257
|
+
all_cols = []
|
|
258
|
+
if has_grouping_id:
|
|
259
|
+
all_cols.append("__GROUPING_ID__")
|
|
260
|
+
for idx in range(len(group_by)):
|
|
261
|
+
all_cols.append(f"__ROW_PATH_{idx}__")
|
|
262
|
+
all_cols.extend(data_columns)
|
|
263
|
+
|
|
264
|
+
grouping_ids = None
|
|
265
|
+
if has_grouping_id:
|
|
266
|
+
grouping_ids = df_slice["__GROUPING_ID__"].to_list()
|
|
267
|
+
|
|
268
|
+
for cidx, col in enumerate(all_cols):
|
|
269
|
+
if cidx == 0 and has_grouping_id:
|
|
270
|
+
continue
|
|
271
|
+
|
|
272
|
+
series = df_slice[col]
|
|
273
|
+
dtype = polars_type_to_psp(series.dtype)
|
|
274
|
+
values = series.to_list()
|
|
275
|
+
|
|
276
|
+
push_col = col
|
|
277
|
+
if is_split_by and not col.startswith("__"):
|
|
278
|
+
push_col = col.replace("_", "|")
|
|
279
|
+
|
|
280
|
+
for ridx, value in enumerate(values):
|
|
281
|
+
if grouping_ids:
|
|
282
|
+
grouping_id = grouping_ids[ridx]
|
|
283
|
+
elif has_group_by:
|
|
284
|
+
grouping_id = 0
|
|
285
|
+
else:
|
|
286
|
+
grouping_id = None
|
|
287
|
+
|
|
288
|
+
if value is not None and isinstance(value, float) and value != value:
|
|
289
|
+
value = None
|
|
290
|
+
|
|
291
|
+
data.set_col(dtype, push_col, ridx, value, grouping_id)
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
################################################################################
|
|
295
|
+
#
|
|
296
|
+
# Polars Utils
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def polars_type_to_psp(dtype):
|
|
300
|
+
"""Convert a Polars `dtype` to a Perspective `ColumnType`."""
|
|
301
|
+
if dtype in (pl.Utf8, pl.String):
|
|
302
|
+
return "string"
|
|
303
|
+
if dtype == pl.Categorical:
|
|
304
|
+
return "string"
|
|
305
|
+
if dtype in (pl.Int8, pl.Int16, pl.Int32, pl.UInt8, pl.UInt16):
|
|
306
|
+
return "integer"
|
|
307
|
+
if dtype in (pl.Int64, pl.UInt64, pl.UInt32, pl.Float32, pl.Float64):
|
|
308
|
+
return "float"
|
|
309
|
+
if dtype == pl.Date:
|
|
310
|
+
return "date"
|
|
311
|
+
if dtype == pl.Boolean:
|
|
312
|
+
return "boolean"
|
|
313
|
+
if isinstance(dtype, pl.Datetime) or dtype == pl.Datetime:
|
|
314
|
+
return "datetime"
|
|
315
|
+
|
|
316
|
+
msg = f"Unknown Polars type '{dtype}'"
|
|
317
|
+
raise ValueError(msg)
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def apply_filters(df, filters):
|
|
321
|
+
"""Apply a list of filter configs to a DataFrame."""
|
|
322
|
+
if not filters:
|
|
323
|
+
return df
|
|
324
|
+
|
|
325
|
+
mask = pl.lit(True)
|
|
326
|
+
for filt in filters:
|
|
327
|
+
col_name = filt[0]
|
|
328
|
+
op = filt[1]
|
|
329
|
+
value = filt[2] if len(filt) > 2 else None
|
|
330
|
+
|
|
331
|
+
if value is None:
|
|
332
|
+
continue
|
|
333
|
+
|
|
334
|
+
col_expr = pl.col(col_name)
|
|
335
|
+
if op == "==":
|
|
336
|
+
mask = mask & (col_expr == value)
|
|
337
|
+
elif op == "!=":
|
|
338
|
+
mask = mask & (col_expr != value)
|
|
339
|
+
elif op == ">":
|
|
340
|
+
mask = mask & (col_expr > value)
|
|
341
|
+
elif op == "<":
|
|
342
|
+
mask = mask & (col_expr < value)
|
|
343
|
+
elif op == ">=":
|
|
344
|
+
mask = mask & (col_expr >= value)
|
|
345
|
+
elif op == "<=":
|
|
346
|
+
mask = mask & (col_expr <= value)
|
|
347
|
+
|
|
348
|
+
return df.filter(mask)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def get_polars_agg_expr(col, agg_name, filter_expr=None):
|
|
352
|
+
"""Convert an aggregate name to a Polars expression."""
|
|
353
|
+
if isinstance(agg_name, list):
|
|
354
|
+
agg_name = agg_name[0]
|
|
355
|
+
if isinstance(agg_name, dict):
|
|
356
|
+
agg_name = "first"
|
|
357
|
+
expr = pl.col(col)
|
|
358
|
+
if filter_expr is not None:
|
|
359
|
+
expr = expr.filter(filter_expr)
|
|
360
|
+
if agg_name in AGG_MAP:
|
|
361
|
+
return AGG_MAP[agg_name](expr)
|
|
362
|
+
|
|
363
|
+
msg = f"Unknown aggregate '{agg_name}'"
|
|
364
|
+
raise ValueError(msg)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def default_aggregate(col_name, df):
|
|
368
|
+
"""Return the default aggregate for a column based on its type."""
|
|
369
|
+
dtype = df[col_name].dtype
|
|
370
|
+
psp_type = polars_type_to_psp(dtype)
|
|
371
|
+
if psp_type in ("integer", "float"):
|
|
372
|
+
return "sum"
|
|
373
|
+
return "count"
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def build_rollup(df, group_by, columns, aggregates, col_alias):
|
|
377
|
+
"""Emulate GROUP BY ROLLUP using multiple group_by operations."""
|
|
378
|
+
n = len(group_by)
|
|
379
|
+
frames = []
|
|
380
|
+
data_columns = [c for c in columns if c not in group_by]
|
|
381
|
+
|
|
382
|
+
for level in range(n + 1):
|
|
383
|
+
num_groups = n - level
|
|
384
|
+
active_groups = group_by[:num_groups]
|
|
385
|
+
|
|
386
|
+
agg_exprs = []
|
|
387
|
+
for col in data_columns:
|
|
388
|
+
agg_name = aggregates.get(col, default_aggregate(col, df))
|
|
389
|
+
agg_exprs.append(get_polars_agg_expr(col, agg_name).alias(col_alias(col)))
|
|
390
|
+
|
|
391
|
+
if active_groups:
|
|
392
|
+
grouped = df.group_by(active_groups, maintain_order=True).agg(agg_exprs)
|
|
393
|
+
else:
|
|
394
|
+
grouped = df.select(agg_exprs)
|
|
395
|
+
|
|
396
|
+
for idx in range(n):
|
|
397
|
+
if idx < num_groups:
|
|
398
|
+
grouped = grouped.with_columns(
|
|
399
|
+
pl.col(group_by[idx]).alias(f"__ROW_PATH_{idx}__")
|
|
400
|
+
)
|
|
401
|
+
else:
|
|
402
|
+
src_dtype = df[group_by[idx]].dtype
|
|
403
|
+
grouped = grouped.with_columns(
|
|
404
|
+
pl.lit(None).cast(src_dtype).alias(f"__ROW_PATH_{idx}__")
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
grouping_id = sum(1 << i for i in range(num_groups, n))
|
|
408
|
+
grouped = grouped.with_columns(
|
|
409
|
+
pl.lit(grouping_id).cast(pl.Int64).alias("__GROUPING_ID__")
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
for gb_col in active_groups:
|
|
413
|
+
if gb_col in grouped.columns:
|
|
414
|
+
grouped = grouped.drop(gb_col)
|
|
415
|
+
|
|
416
|
+
frames.append(grouped)
|
|
417
|
+
|
|
418
|
+
result = pl.concat(frames, how="diagonal")
|
|
419
|
+
path_cols = [f"__ROW_PATH_{i}__" for i in range(n)]
|
|
420
|
+
data_col_aliases = [col_alias(c) for c in data_columns]
|
|
421
|
+
final_order = ["__GROUPING_ID__"] + path_cols + data_col_aliases
|
|
422
|
+
result = result.select([c for c in final_order if c in result.columns])
|
|
423
|
+
return result
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def build_flat_group_by(df, group_by, columns, aggregates, col_alias):
|
|
427
|
+
"""Build a simple GROUP BY (no rollup) - only leaf-level rows."""
|
|
428
|
+
n = len(group_by)
|
|
429
|
+
data_columns = [c for c in columns if c not in group_by]
|
|
430
|
+
|
|
431
|
+
agg_exprs = []
|
|
432
|
+
for col in data_columns:
|
|
433
|
+
agg_name = aggregates.get(col, default_aggregate(col, df))
|
|
434
|
+
agg_exprs.append(get_polars_agg_expr(col, agg_name).alias(col_alias(col)))
|
|
435
|
+
|
|
436
|
+
grouped = df.group_by(group_by, maintain_order=True).agg(agg_exprs)
|
|
437
|
+
|
|
438
|
+
for idx in range(n):
|
|
439
|
+
grouped = grouped.with_columns(
|
|
440
|
+
pl.col(group_by[idx]).alias(f"__ROW_PATH_{idx}__")
|
|
441
|
+
)
|
|
442
|
+
|
|
443
|
+
for gb_col in group_by:
|
|
444
|
+
if gb_col in grouped.columns:
|
|
445
|
+
grouped = grouped.drop(gb_col)
|
|
446
|
+
|
|
447
|
+
path_cols = [f"__ROW_PATH_{i}__" for i in range(n)]
|
|
448
|
+
data_col_aliases = [col_alias(c) for c in data_columns]
|
|
449
|
+
final_order = path_cols + data_col_aliases
|
|
450
|
+
result = grouped.select([c for c in final_order if c in grouped.columns])
|
|
451
|
+
return result.sort(path_cols)
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def build_total(df, columns, aggregates, col_alias):
|
|
455
|
+
"""Build a single total row aggregating the entire dataset."""
|
|
456
|
+
agg_exprs = []
|
|
457
|
+
for col in columns:
|
|
458
|
+
agg_name = aggregates.get(col, default_aggregate(col, df))
|
|
459
|
+
agg_exprs.append(get_polars_agg_expr(col, agg_name).alias(col_alias(col)))
|
|
460
|
+
return df.select(agg_exprs)
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def build_split_by_total(df, split_by, columns, aggregates, col_alias):
|
|
464
|
+
"""Build a single total row with split_by (pivot) columns."""
|
|
465
|
+
split_col = split_by[0]
|
|
466
|
+
data_columns = [c for c in columns if c not in split_by]
|
|
467
|
+
split_values = sorted(df[split_col].unique().to_list())
|
|
468
|
+
|
|
469
|
+
agg_exprs = []
|
|
470
|
+
for sv in split_values:
|
|
471
|
+
filter_expr = pl.col(split_col) == sv
|
|
472
|
+
for dc in data_columns:
|
|
473
|
+
agg_name = aggregates.get(dc, default_aggregate(dc, df))
|
|
474
|
+
col_name = f"{sv}_{col_alias(dc)}"
|
|
475
|
+
agg_exprs.append(
|
|
476
|
+
get_polars_agg_expr(dc, agg_name, filter_expr=filter_expr).alias(
|
|
477
|
+
col_name
|
|
478
|
+
)
|
|
479
|
+
)
|
|
480
|
+
|
|
481
|
+
return df.select(agg_exprs)
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def build_split_by_grouped_flat(df, group_by, split_by, columns, aggregates, col_alias):
|
|
485
|
+
"""Build a flat grouped view with split_by (pivot) columns - no rollup rows."""
|
|
486
|
+
n = len(group_by)
|
|
487
|
+
split_col = split_by[0]
|
|
488
|
+
data_columns = [c for c in columns if c not in group_by and c not in split_by]
|
|
489
|
+
split_values = sorted(df[split_col].unique().to_list())
|
|
490
|
+
|
|
491
|
+
agg_exprs = []
|
|
492
|
+
for sv in split_values:
|
|
493
|
+
filter_expr = pl.col(split_col) == sv
|
|
494
|
+
for dc in data_columns:
|
|
495
|
+
agg_name = aggregates.get(dc, default_aggregate(dc, df))
|
|
496
|
+
col_name = f"{sv}_{col_alias(dc)}"
|
|
497
|
+
agg_exprs.append(
|
|
498
|
+
get_polars_agg_expr(dc, agg_name, filter_expr=filter_expr).alias(
|
|
499
|
+
col_name
|
|
500
|
+
)
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
for dc in data_columns:
|
|
504
|
+
agg_name = aggregates.get(dc, default_aggregate(dc, df))
|
|
505
|
+
agg_exprs.append(get_polars_agg_expr(dc, agg_name).alias(f"__SORT_{dc}__"))
|
|
506
|
+
|
|
507
|
+
grouped = df.group_by(group_by, maintain_order=True).agg(agg_exprs)
|
|
508
|
+
|
|
509
|
+
for idx in range(n):
|
|
510
|
+
grouped = grouped.with_columns(
|
|
511
|
+
pl.col(group_by[idx]).alias(f"__ROW_PATH_{idx}__")
|
|
512
|
+
)
|
|
513
|
+
|
|
514
|
+
for gb_col in group_by:
|
|
515
|
+
if gb_col in grouped.columns:
|
|
516
|
+
grouped = grouped.drop(gb_col)
|
|
517
|
+
|
|
518
|
+
path_cols = [f"__ROW_PATH_{i}__" for i in range(n)]
|
|
519
|
+
data_col_names = []
|
|
520
|
+
for sv in split_values:
|
|
521
|
+
for dc in data_columns:
|
|
522
|
+
data_col_names.append(f"{sv}_{col_alias(dc)}")
|
|
523
|
+
sort_col_names = [f"__SORT_{dc}__" for dc in data_columns]
|
|
524
|
+
final_order = path_cols + data_col_names + sort_col_names
|
|
525
|
+
result = grouped.select([c for c in final_order if c in grouped.columns])
|
|
526
|
+
return result.sort(path_cols)
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
def apply_sort_grouped(df, sort_config, group_by, col_alias):
|
|
530
|
+
"""Apply sort to a ROLLUP result DataFrame."""
|
|
531
|
+
n = len(group_by)
|
|
532
|
+
|
|
533
|
+
sort_cols = []
|
|
534
|
+
sort_desc = []
|
|
535
|
+
for entry in sort_config:
|
|
536
|
+
col = entry[0]
|
|
537
|
+
direction = entry[1]
|
|
538
|
+
if direction != "none":
|
|
539
|
+
aliased = col_alias(col)
|
|
540
|
+
if aliased in df.columns:
|
|
541
|
+
sort_cols.append(aliased)
|
|
542
|
+
sort_desc.append(direction in ("desc", "col desc"))
|
|
543
|
+
|
|
544
|
+
if not sort_cols:
|
|
545
|
+
# Default: tree order by row path, nulls first
|
|
546
|
+
path_cols = [f"__ROW_PATH_{i}__" for i in range(n)]
|
|
547
|
+
return df.sort(path_cols, descending=[False] * n, nulls_last=False)
|
|
548
|
+
|
|
549
|
+
# With explicit sort: grand total first, then rest sorted
|
|
550
|
+
is_total = pl.lit(True)
|
|
551
|
+
for i in range(n):
|
|
552
|
+
is_total = is_total & pl.col(f"__ROW_PATH_{i}__").is_null()
|
|
553
|
+
|
|
554
|
+
total_row = df.filter(is_total)
|
|
555
|
+
rest = df.filter(~is_total)
|
|
556
|
+
rest = rest.sort(sort_cols, descending=sort_desc)
|
|
557
|
+
return pl.concat([total_row, rest])
|
|
558
|
+
|
|
559
|
+
|
|
560
|
+
def apply_sort_split_by_flat(df, sort_config, columns, group_by, split_by):
|
|
561
|
+
"""Apply sort to a flat split_by grouped DataFrame using __SORT__ columns."""
|
|
562
|
+
data_columns = [c for c in columns if c not in group_by and c not in split_by]
|
|
563
|
+
sort_cols = []
|
|
564
|
+
sort_desc = []
|
|
565
|
+
for entry in sort_config:
|
|
566
|
+
col = entry[0]
|
|
567
|
+
direction = entry[1]
|
|
568
|
+
if direction != "none":
|
|
569
|
+
sort_name = f"__SORT_{col}__"
|
|
570
|
+
if sort_name in df.columns:
|
|
571
|
+
sort_cols.append(sort_name)
|
|
572
|
+
sort_desc.append(direction in ("desc", "col desc"))
|
|
573
|
+
|
|
574
|
+
if sort_cols:
|
|
575
|
+
df = df.sort(sort_cols, descending=sort_desc)
|
|
576
|
+
|
|
577
|
+
drop_cols = [
|
|
578
|
+
f"__SORT_{dc}__" for dc in data_columns if f"__SORT_{dc}__" in df.columns
|
|
579
|
+
]
|
|
580
|
+
if drop_cols:
|
|
581
|
+
df = df.drop(drop_cols)
|
|
582
|
+
return df
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def apply_sort_flat(df, sort_config, col_alias):
|
|
586
|
+
"""Apply sort to a flat (non-grouped) DataFrame."""
|
|
587
|
+
if not sort_config:
|
|
588
|
+
return df
|
|
589
|
+
|
|
590
|
+
sort_cols = []
|
|
591
|
+
sort_descending = []
|
|
592
|
+
for sort_entry in sort_config:
|
|
593
|
+
col = sort_entry[0]
|
|
594
|
+
direction = sort_entry[1]
|
|
595
|
+
if direction != "none":
|
|
596
|
+
aliased = col_alias(col)
|
|
597
|
+
if aliased in df.columns:
|
|
598
|
+
sort_cols.append(aliased)
|
|
599
|
+
sort_descending.append(direction in ("desc", "col desc"))
|
|
600
|
+
|
|
601
|
+
if sort_cols:
|
|
602
|
+
return df.sort(sort_cols, descending=sort_descending)
|
|
603
|
+
return df
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
def compute_view_schema(df):
|
|
607
|
+
"""Compute the Perspective schema for a view DataFrame."""
|
|
608
|
+
schema = {}
|
|
609
|
+
for col_name, dtype in df.schema.items():
|
|
610
|
+
if col_name.startswith("__"):
|
|
611
|
+
continue
|
|
612
|
+
schema[col_name] = polars_type_to_psp(dtype)
|
|
613
|
+
return schema
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def build_split_by_grouped(df, group_by, split_by, columns, aggregates, col_alias):
|
|
617
|
+
"""Build a grouped rollup with split_by (pivot) columns."""
|
|
618
|
+
n = len(group_by)
|
|
619
|
+
split_col = split_by[0]
|
|
620
|
+
data_columns = [c for c in columns if c not in group_by and c not in split_by]
|
|
621
|
+
split_values = sorted(df[split_col].unique().to_list())
|
|
622
|
+
|
|
623
|
+
frames = []
|
|
624
|
+
for level in range(n + 1):
|
|
625
|
+
num_groups = n - level
|
|
626
|
+
active_groups = group_by[:num_groups]
|
|
627
|
+
|
|
628
|
+
agg_exprs = []
|
|
629
|
+
for sv in split_values:
|
|
630
|
+
filter_expr = pl.col(split_col) == sv
|
|
631
|
+
for dc in data_columns:
|
|
632
|
+
agg_name = aggregates.get(dc, default_aggregate(dc, df))
|
|
633
|
+
col_name = f"{sv}_{col_alias(dc)}"
|
|
634
|
+
agg_exprs.append(
|
|
635
|
+
get_polars_agg_expr(dc, agg_name, filter_expr=filter_expr).alias(
|
|
636
|
+
col_name
|
|
637
|
+
)
|
|
638
|
+
)
|
|
639
|
+
|
|
640
|
+
if active_groups:
|
|
641
|
+
grouped = df.group_by(active_groups, maintain_order=True).agg(agg_exprs)
|
|
642
|
+
else:
|
|
643
|
+
grouped = df.select(agg_exprs)
|
|
644
|
+
|
|
645
|
+
for idx in range(n):
|
|
646
|
+
if idx < num_groups:
|
|
647
|
+
grouped = grouped.with_columns(
|
|
648
|
+
pl.col(group_by[idx]).alias(f"__ROW_PATH_{idx}__")
|
|
649
|
+
)
|
|
650
|
+
else:
|
|
651
|
+
src_dtype = df[group_by[idx]].dtype
|
|
652
|
+
grouped = grouped.with_columns(
|
|
653
|
+
pl.lit(None).cast(src_dtype).alias(f"__ROW_PATH_{idx}__")
|
|
654
|
+
)
|
|
655
|
+
|
|
656
|
+
grouping_id = sum(1 << i for i in range(num_groups, n))
|
|
657
|
+
grouped = grouped.with_columns(
|
|
658
|
+
pl.lit(grouping_id).cast(pl.Int64).alias("__GROUPING_ID__")
|
|
659
|
+
)
|
|
660
|
+
|
|
661
|
+
for gb_col in active_groups:
|
|
662
|
+
if gb_col in grouped.columns:
|
|
663
|
+
grouped = grouped.drop(gb_col)
|
|
664
|
+
|
|
665
|
+
frames.append(grouped)
|
|
666
|
+
|
|
667
|
+
result = pl.concat(frames, how="diagonal")
|
|
668
|
+
path_cols = [f"__ROW_PATH_{i}__" for i in range(n)]
|
|
669
|
+
data_col_names = []
|
|
670
|
+
for sv in split_values:
|
|
671
|
+
for dc in data_columns:
|
|
672
|
+
data_col_names.append(f"{sv}_{col_alias(dc)}")
|
|
673
|
+
final_order = ["__GROUPING_ID__"] + path_cols + data_col_names
|
|
674
|
+
result = result.select([c for c in final_order if c in result.columns])
|
|
675
|
+
return result
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def build_split_by_flat(df, split_by, columns, col_alias):
|
|
679
|
+
"""Build a flat (non-grouped) split_by view."""
|
|
680
|
+
split_col = split_by[0]
|
|
681
|
+
data_columns = [c for c in columns if c not in split_by]
|
|
682
|
+
split_values = sorted(df[split_col].unique().to_list())
|
|
683
|
+
|
|
684
|
+
exprs = []
|
|
685
|
+
for sv in split_values:
|
|
686
|
+
for dc in data_columns:
|
|
687
|
+
col_name = f"{sv}_{col_alias(dc)}"
|
|
688
|
+
exprs.append(
|
|
689
|
+
pl.when(pl.col(split_col) == sv)
|
|
690
|
+
.then(pl.col(dc))
|
|
691
|
+
.otherwise(None)
|
|
692
|
+
.alias(col_name)
|
|
693
|
+
)
|
|
694
|
+
|
|
695
|
+
return df.select(exprs)
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
def parse_expression(expr_str):
|
|
699
|
+
"""Parse a Perspective expression string into a Polars expression."""
|
|
700
|
+
pattern = r'"([^"]*)"'
|
|
701
|
+
parts = []
|
|
702
|
+
last_end = 0
|
|
703
|
+
for match in re.finditer(pattern, expr_str):
|
|
704
|
+
parts.append(expr_str[last_end : match.start()])
|
|
705
|
+
col_name = match.group(1)
|
|
706
|
+
parts.append(f'pl.col("{col_name}")')
|
|
707
|
+
last_end = match.end()
|
|
708
|
+
parts.append(expr_str[last_end:])
|
|
709
|
+
polars_expr_str = "".join(parts)
|
|
710
|
+
return eval(polars_expr_str, {"pl": pl, "__builtins__": {}})
|