py-tbparse 0.1.1a0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- py_tbparse-0.1.1a0.dist-info/METADATA +168 -0
- py_tbparse-0.1.1a0.dist-info/RECORD +26 -0
- py_tbparse-0.1.1a0.dist-info/WHEEL +5 -0
- py_tbparse-0.1.1a0.dist-info/entry_points.txt +3 -0
- py_tbparse-0.1.1a0.dist-info/licenses/LICENSE +25 -0
- py_tbparse-0.1.1a0.dist-info/top_level.txt +1 -0
- twbparser_py/__init__.py +54 -0
- twbparser_py/_clean.py +87 -0
- twbparser_py/_tables.py +97 -0
- twbparser_py/_xml.py +172 -0
- twbparser_py/batch.py +62 -0
- twbparser_py/calculated_fields.py +113 -0
- twbparser_py/cli.py +207 -0
- twbparser_py/dashboards.py +101 -0
- twbparser_py/datasources.py +257 -0
- twbparser_py/diff.py +86 -0
- twbparser_py/fields.py +148 -0
- twbparser_py/graph.py +81 -0
- twbparser_py/joins.py +99 -0
- twbparser_py/parser.py +234 -0
- twbparser_py/published.py +52 -0
- twbparser_py/py.typed +0 -0
- twbparser_py/relationships.py +175 -0
- twbparser_py/sql.py +76 -0
- twbparser_py/validators.py +99 -0
- twbparser_py/webgui.py +398 -0
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: py-tbparse
|
|
3
|
+
Version: 0.1.1a0
|
|
4
|
+
Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
|
|
5
|
+
Author: DDSNA
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: tableau,twb,twbx,workbook,parser,pandas
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
19
|
+
Classifier: Topic :: Office/Business
|
|
20
|
+
Requires-Python: >=3.9
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Requires-Dist: lxml>=4.9
|
|
24
|
+
Requires-Dist: pandas>=1.5
|
|
25
|
+
Provides-Extra: test
|
|
26
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
27
|
+
Provides-Extra: browser
|
|
28
|
+
Requires-Dist: playwright>=1.40; extra == "browser"
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
31
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
34
|
+
# twbparser-py
|
|
35
|
+
|
|
36
|
+
A native Python port of the [`twbparser`](https://github.com/PrigasG/twbparser)
|
|
37
|
+
R package: parses Tableau `.twb`/`.twbx` workbook files into `pandas`
|
|
38
|
+
DataFrames. No R runtime required — pure `lxml` XML parsing.
|
|
39
|
+
|
|
40
|
+
This is a v1 subset covering the parser's core: workbook loading,
|
|
41
|
+
datasources, parameters, fields, calculated fields, joins, relationships
|
|
42
|
+
(legacy and 2020.2+), inferred relationships, dashboards, relationship
|
|
43
|
+
validation, custom/initial SQL, and published-source detection.
|
|
44
|
+
Formatting/tooltips/colors/axes/sorts, dashboard layout/actions,
|
|
45
|
+
analytics helpers (calc complexity, field usage, replication brief), and
|
|
46
|
+
the Shiny-inspector equivalent are not yet ported.
|
|
47
|
+
|
|
48
|
+
Beyond the R original, this port also adds a few Python-native extras: a
|
|
49
|
+
Graphviz DOT export of the relationship graph, a workbook-to-workbook
|
|
50
|
+
diff, folder/batch analysis across many workbooks, and Jupyter rich
|
|
51
|
+
display (`_repr_html_`).
|
|
52
|
+
|
|
53
|
+
## Install
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
pip install -e .
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Usage
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
from twbparser_py import TwbParser
|
|
63
|
+
|
|
64
|
+
p = TwbParser("workbook.twb") # or .twbx
|
|
65
|
+
p.get_datasources()
|
|
66
|
+
p.get_fields()
|
|
67
|
+
p.get_calculated_fields()
|
|
68
|
+
p.get_joins()
|
|
69
|
+
p.get_relationships()
|
|
70
|
+
p.get_inferred_relationships()
|
|
71
|
+
p.get_dashboards()
|
|
72
|
+
p.get_dashboard_sheets()
|
|
73
|
+
p.get_custom_sql()
|
|
74
|
+
p.get_initial_sql()
|
|
75
|
+
p.get_published_refs()
|
|
76
|
+
p.get_relationship_graph_dot() # Graphviz DOT string
|
|
77
|
+
p.validate()
|
|
78
|
+
p.get_overview()
|
|
79
|
+
p # in Jupyter: renders get_overview() via _repr_html_
|
|
80
|
+
|
|
81
|
+
from twbparser_py import diff_workbooks, scan_folder
|
|
82
|
+
|
|
83
|
+
diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
|
|
84
|
+
scan_folder("./workbooks", table="datasources") # one row per workbook x datasource
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
### CLI
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
twbparser workbook.twb # overview (default table)
|
|
91
|
+
twbparser workbook.twb tables # list available tables
|
|
92
|
+
twbparser workbook.twb calculated-fields # print a table
|
|
93
|
+
twbparser workbook.twb fields --format csv -o fields.csv
|
|
94
|
+
twbparser workbook.twbx dashboard-sheets --dashboard "Sales Overview"
|
|
95
|
+
twbparser workbook.twb validate # exit code 2 if invalid
|
|
96
|
+
twbparser workbook.twb graph --include-inferred > relationships.dot
|
|
97
|
+
twbparser diff old.twb new.twb datasources # row-level added/removed
|
|
98
|
+
twbparser batch ./workbooks datasources # one table, every workbook in a folder
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`,
|
|
102
|
+
`calculated-fields`, `joins`, `relations`, `relationships`,
|
|
103
|
+
`inferred-relationships`, `dashboards`, `dashboard-sheets`,
|
|
104
|
+
`custom-sql`, `initial-sql`, `published-refs`. `--format` is `table`
|
|
105
|
+
(default), `csv`, or `json` (`graph` always prints Graphviz DOT text
|
|
106
|
+
regardless of `--format`). `diff`/`batch` accept most of the same table
|
|
107
|
+
names, minus `graph`/`validate`/`tables`.
|
|
108
|
+
|
|
109
|
+
### GUI
|
|
110
|
+
|
|
111
|
+
A local, browser-based GUI — standard library only (`http.server` +
|
|
112
|
+
vanilla JS), no GUI toolkit or extra dependency required:
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
twbparser-gui workbook.twb # opens your default browser
|
|
116
|
+
twbparser-gui # opens with an empty path field; paste one and click Load
|
|
117
|
+
twbparser-gui --no-browser --port 8765 # just run the server, e.g. for a headless box
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Pick a table from the dropdown, filter `dashboard-sheets` by dashboard,
|
|
121
|
+
toggle "include Parameters" for `calculated-fields`, pick `graph` to
|
|
122
|
+
preview/export a Graphviz DOT digraph of the relationships, and export
|
|
123
|
+
any tabular view as CSV. All state lives server-side in memory for the
|
|
124
|
+
life of the process — it's a single-user local tool, not something to
|
|
125
|
+
expose on a shared network.
|
|
126
|
+
|
|
127
|
+
## Testing
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
pip install -e ".[test]"
|
|
131
|
+
pytest
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Fixtures in `tests/fixtures/` are the same tiny sample workbooks used by
|
|
135
|
+
the original R package's test suite (`inst/extdata/`).
|
|
136
|
+
|
|
137
|
+
The GUI additionally has end-to-end tests that drive the page in a real
|
|
138
|
+
headless Chromium (via Playwright) and fail on any uncaught JavaScript
|
|
139
|
+
error. They're opt-in — without the browser installed they skip and the
|
|
140
|
+
rest of the suite runs normally:
|
|
141
|
+
|
|
142
|
+
```bash
|
|
143
|
+
pip install -e ".[test,browser]"
|
|
144
|
+
playwright install chromium # add --with-deps if you have root
|
|
145
|
+
./scripts/setup-browser-libs.sh # no-root alternative to --with-deps
|
|
146
|
+
pytest tests/test_gui_browser.py
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## Background
|
|
150
|
+
|
|
151
|
+
`.twb` is plain XML, and `.twbx` is just a zip wrapper around one, which
|
|
152
|
+
is why parsing it natively in Python — no R, no reverse-engineering —
|
|
153
|
+
was tractable at this scale. Power BI's equivalent format, `.pbix`, is a
|
|
154
|
+
binary container built around the proprietary VertiPaq storage engine,
|
|
155
|
+
which is why reading it programmatically needed dedicated
|
|
156
|
+
reverse-engineering projects like
|
|
157
|
+
[PBIXRay](https://github.com/Hugoberry/pbixray) and
|
|
158
|
+
[pbi-tools](https://github.com/pbi-tools/pbi-tools). The comparison
|
|
159
|
+
isn't one-sided, though: Tableau's own official Python tooling for
|
|
160
|
+
*server* automation
|
|
161
|
+
([`tableauserverclient`](https://pypi.org/project/tableauserverclient/),
|
|
162
|
+
`tabcmd`) is more mature and more open than anything Microsoft ships for
|
|
163
|
+
Power BI's REST API.
|
|
164
|
+
|
|
165
|
+
## Credit
|
|
166
|
+
|
|
167
|
+
Ported from the R implementation by George Arthur
|
|
168
|
+
([`PrigasG/twbparser`](https://github.com/PrigasG/twbparser)), MIT licensed.
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
py_tbparse-0.1.1a0.dist-info/licenses/LICENSE,sha256=y8NSwuzSKYNP-QLxH5gd8cYj3m_yNcufgRWRjaC-NCM,1252
|
|
2
|
+
twbparser_py/__init__.py,sha256=Mj5SX7NFlXweWg23nmV9sI6sz34-YtfbAsNlBFAFSps,1856
|
|
3
|
+
twbparser_py/_clean.py,sha256=moCudA2sxmlVI5HBM5EI8BMus_fcMKPOVomeycbOkVY,2313
|
|
4
|
+
twbparser_py/_tables.py,sha256=r-4rXT7Z71XQxn1oz75DC-4wIh2ejTi3UkwgTd2ahwI,2550
|
|
5
|
+
twbparser_py/_xml.py,sha256=C_EjP58cav713GnWrfENga9UZqCVnBRL51H0WcSAcMw,5568
|
|
6
|
+
twbparser_py/batch.py,sha256=WetmMfNYpSGP8cCFCKsP6wYG-lq-wAO-ZS4bCd4APUA,1804
|
|
7
|
+
twbparser_py/calculated_fields.py,sha256=tRo4zCRTJ-4LiMp9QSlq1OTN-Co-snNjLaTALYfvQDo,3859
|
|
8
|
+
twbparser_py/cli.py,sha256=awMN7NnkNX-fQpmnN-Rf3g9KyGQKUD-YLx8t2ztK6I8,7181
|
|
9
|
+
twbparser_py/dashboards.py,sha256=g3fQrqLIc88J6ug3PrCUIXoP8TJV8VZhlG6Av8SM4_U,3282
|
|
10
|
+
twbparser_py/datasources.py,sha256=c924vxYR9_lQKX0GRNQFGuZ3bHXpxQZ7V2Q9PQUoxHA,8790
|
|
11
|
+
twbparser_py/diff.py,sha256=CB1WJUF7cb7H7KOZPObS__SELHmlbL1J3QOb5aHMs1g,3303
|
|
12
|
+
twbparser_py/fields.py,sha256=MlWpdXessuW_mZ8qLZFl6d9-G9coMRz8rxiakEvLKHc,5378
|
|
13
|
+
twbparser_py/graph.py,sha256=sMbh4wayhRxpA7hBNBGiyOF4lGSd_cEKW8sXjdwfw8c,2955
|
|
14
|
+
twbparser_py/joins.py,sha256=Df0O5edFQ9pXjGPgrT8Q4TPOzDP6qNJm0nJ3cGAPewI,3503
|
|
15
|
+
twbparser_py/parser.py,sha256=Ip4hPMCzi8vWeCYCG45kIXt8y6SL0RSngqsYBs4NvQk,8967
|
|
16
|
+
twbparser_py/published.py,sha256=vIFKBI4pw-m9vqXFa5dqY2nGJWsTjN6pAhysj8D2RLI,1472
|
|
17
|
+
twbparser_py/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
18
|
+
twbparser_py/relationships.py,sha256=uj2k67NEeNG-dCs-uZsVc8ghCtPFeK-Zbfnc8q5KRlk,5669
|
|
19
|
+
twbparser_py/sql.py,sha256=lQonS9GXSAU7pR7yhjCK0WdGBtMTP3bsCXdDkIeWVwI,2443
|
|
20
|
+
twbparser_py/validators.py,sha256=1rG2-1hORa7joNy7cr3PhVfQsASZZjV30PfhItyCSN8,3492
|
|
21
|
+
twbparser_py/webgui.py,sha256=flrdhxf-s0mzOwSKYHIl-r2CZXDYuqtyosUk8-71NdI,14040
|
|
22
|
+
py_tbparse-0.1.1a0.dist-info/METADATA,sha256=OQU06QCv-7AlG9EzCOwVXfWbX1S9yeRW1YqWstv04XU,6402
|
|
23
|
+
py_tbparse-0.1.1a0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
24
|
+
py_tbparse-0.1.1a0.dist-info/entry_points.txt,sha256=VIH-RDvE5jddcwiDat_CO4nQVLcnTFxfgx4D4vQS39A,93
|
|
25
|
+
py_tbparse-0.1.1a0.dist-info/top_level.txt,sha256=ZEo8_tUfoJEe1tKN5aJCS-mvL_3QAnu8qfELdfaDSnQ,13
|
|
26
|
+
py_tbparse-0.1.1a0.dist-info/RECORD,,
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
This project is a Python port of logic originally implemented in the R
|
|
4
|
+
package "twbparser" (https://github.com/PrigasG/twbparser),
|
|
5
|
+
Copyright (c) 2025 George Arthur.
|
|
6
|
+
|
|
7
|
+
Copyright (c) 2026 the twbparser-py contributors
|
|
8
|
+
|
|
9
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
10
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
11
|
+
in the Software without restriction, including without limitation the rights
|
|
12
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
13
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
14
|
+
furnished to do so, subject to the following conditions:
|
|
15
|
+
|
|
16
|
+
The above copyright notice and this permission notice shall be included in all
|
|
17
|
+
copies or substantial portions of the Software.
|
|
18
|
+
|
|
19
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
20
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
21
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
22
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
23
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
24
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
25
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
twbparser_py
|
twbparser_py/__init__.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""twbparser_py: a native Python port of the twbparser R package.
|
|
2
|
+
|
|
3
|
+
Parses Tableau .twb/.twbx workbook files into pandas DataFrames. Ported
|
|
4
|
+
from https://github.com/PrigasG/twbparser (MIT licensed).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .batch import scan_folder
|
|
8
|
+
from .calculated_fields import extract_calculated_fields, extract_raw_fields
|
|
9
|
+
from .dashboards import dashboard_sheets, list_dashboards
|
|
10
|
+
from .datasources import extract_datasource_details, extract_named_connections, extract_parameters
|
|
11
|
+
from .diff import diff_tables, diff_workbooks
|
|
12
|
+
from .fields import extract_columns_with_table_source, infer_implicit_relationships
|
|
13
|
+
from .graph import to_dot
|
|
14
|
+
from .joins import extract_joins
|
|
15
|
+
from .parser import TwbParser
|
|
16
|
+
from .published import extract_published_refs
|
|
17
|
+
from .relationships import extract_relations, extract_relationships
|
|
18
|
+
from .sql import extract_custom_sql, extract_initial_sql
|
|
19
|
+
from .validators import validate_relationships
|
|
20
|
+
from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"TwbParser",
|
|
24
|
+
"extract_calculated_fields",
|
|
25
|
+
"extract_raw_fields",
|
|
26
|
+
"dashboard_sheets",
|
|
27
|
+
"list_dashboards",
|
|
28
|
+
"extract_datasource_details",
|
|
29
|
+
"extract_named_connections",
|
|
30
|
+
"extract_parameters",
|
|
31
|
+
"extract_columns_with_table_source",
|
|
32
|
+
"infer_implicit_relationships",
|
|
33
|
+
"extract_joins",
|
|
34
|
+
"to_dot",
|
|
35
|
+
"extract_relations",
|
|
36
|
+
"extract_relationships",
|
|
37
|
+
"extract_custom_sql",
|
|
38
|
+
"extract_initial_sql",
|
|
39
|
+
"extract_published_refs",
|
|
40
|
+
"validate_relationships",
|
|
41
|
+
"extract_twb_from_twbx",
|
|
42
|
+
"twbx_extract_files",
|
|
43
|
+
"twbx_list",
|
|
44
|
+
"diff_tables",
|
|
45
|
+
"diff_workbooks",
|
|
46
|
+
"scan_folder",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
try:
|
|
50
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
51
|
+
|
|
52
|
+
__version__ = version("twbparser-py")
|
|
53
|
+
except PackageNotFoundError: # pragma: no cover - not installed, e.g. running from source
|
|
54
|
+
__version__ = "0.0.0+unknown"
|
twbparser_py/_clean.py
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Name-cleaning helpers, ported from twbparser's R/utils.R."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
_TRAILING_HEX32 = re.compile(r"_[0-9A-Fa-f]{32}$")
|
|
11
|
+
_LEADING_BRACKET_PREFIX = re.compile(r"^\[.*?\]\.")
|
|
12
|
+
_BRACKETS = re.compile(r"[\[\]]")
|
|
13
|
+
_DERIVATION_WRAPPER = re.compile(r"^[a-z]+:(.+):[a-z]{1,3}$")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def clean_table(x: Optional[str]) -> Optional[str]:
|
|
17
|
+
"""Port of `.twb_clean_table`.
|
|
18
|
+
|
|
19
|
+
Drops `[Extract].`/`[Connection].` prefixes, strips `[]`, and removes
|
|
20
|
+
Tableau's trailing 32-char hex suffix.
|
|
21
|
+
"""
|
|
22
|
+
if x is None:
|
|
23
|
+
return None
|
|
24
|
+
s = str(x)
|
|
25
|
+
s = _LEADING_BRACKET_PREFIX.sub("", s)
|
|
26
|
+
s = _BRACKETS.sub("", s)
|
|
27
|
+
s = _TRAILING_HEX32.sub("", s)
|
|
28
|
+
s = s.strip()
|
|
29
|
+
return s if s else None
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def clean_field(x: Optional[str]) -> Optional[str]:
|
|
33
|
+
"""Port of `.twb_clean_field`.
|
|
34
|
+
|
|
35
|
+
Strips `[]`, takes the last dot-separated token, then unwraps a
|
|
36
|
+
Tableau column-instance wrapper like `none:Category:nk` -> `Category`.
|
|
37
|
+
"""
|
|
38
|
+
if x is None:
|
|
39
|
+
return None
|
|
40
|
+
s = str(x)
|
|
41
|
+
s = _BRACKETS.sub("", s)
|
|
42
|
+
|
|
43
|
+
parts = [p for p in s.split(".") if p]
|
|
44
|
+
token = parts[-1] if parts else None
|
|
45
|
+
if token is None:
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
m = _DERIVATION_WRAPPER.match(token)
|
|
49
|
+
if m:
|
|
50
|
+
return m.group(1)
|
|
51
|
+
return token
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def is_missing(x) -> bool:
|
|
55
|
+
"""True for any "missing" scalar (`None`, float `NaN`, `NaT`, `pd.NA`,
|
|
56
|
+
...) -- not a port of an R function, but a shared helper so callers
|
|
57
|
+
(`diff.py`, `validators.py`) don't each reimplement `pd.isna()`'s
|
|
58
|
+
edge cases (e.g. it raises on array-likes) slightly differently."""
|
|
59
|
+
if x is None:
|
|
60
|
+
return True
|
|
61
|
+
try:
|
|
62
|
+
return bool(pd.isna(x))
|
|
63
|
+
except (TypeError, ValueError):
|
|
64
|
+
return False
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def attr_safe_get(attrs: dict, name: str, default=None):
|
|
68
|
+
"""Port of `attr_safe_get`."""
|
|
69
|
+
if attrs is None:
|
|
70
|
+
return default
|
|
71
|
+
return attrs.get(name, default)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def strip_brackets(x: Optional[str]) -> Optional[str]:
|
|
75
|
+
"""Port of `.strip_brackets`."""
|
|
76
|
+
if x is None:
|
|
77
|
+
return None
|
|
78
|
+
return _BRACKETS.sub("", x)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def basename_safe(x: Optional[str], fallback: str = "<unknown>") -> str:
|
|
82
|
+
"""Port of `basename_safe`."""
|
|
83
|
+
if not x:
|
|
84
|
+
return fallback
|
|
85
|
+
import os
|
|
86
|
+
|
|
87
|
+
return os.path.basename(x)
|
twbparser_py/_tables.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Shared table registry used by both the CLI and the web GUI.
|
|
2
|
+
|
|
3
|
+
Keeping this in one place means the CLI and GUI can never drift out of
|
|
4
|
+
sync on which tables exist or how their optional parameters
|
|
5
|
+
(`dashboard`, `include_parameters`) are threaded through to `TwbParser`.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Callable
|
|
11
|
+
|
|
12
|
+
import pandas as pd
|
|
13
|
+
|
|
14
|
+
from .parser import TwbParser
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _overview(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
18
|
+
return p.get_overview()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _datasources(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
22
|
+
return p.get_datasources()
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _parameters(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
26
|
+
return p.get_parameters()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _fields(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
30
|
+
return p.get_fields()
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _raw_fields(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
34
|
+
return p.get_raw_fields()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _calculated_fields(p: TwbParser, include_parameters: bool = False, **_kw) -> pd.DataFrame:
|
|
38
|
+
return p.get_calculated_fields(include_parameters=include_parameters)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _joins(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
42
|
+
return p.get_joins()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _relations(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
46
|
+
return p.get_relations()
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _relationships(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
50
|
+
return p.get_relationships()
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _inferred_relationships(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
54
|
+
return p.get_inferred_relationships()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _dashboards(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
58
|
+
return p.get_dashboards()
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _dashboard_sheets(p: TwbParser, dashboard: str | None = None, **_kw) -> pd.DataFrame:
|
|
62
|
+
return p.get_dashboard_sheets(dashboard=dashboard or None)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _custom_sql(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
66
|
+
return p.get_custom_sql()
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _initial_sql(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
70
|
+
return p.get_initial_sql()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _published_refs(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
74
|
+
return p.get_published_refs()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# Order here is display order in both the CLI's `tables` listing and the
|
|
78
|
+
# GUI's table dropdown.
|
|
79
|
+
TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
|
|
80
|
+
"overview": _overview,
|
|
81
|
+
"datasources": _datasources,
|
|
82
|
+
"parameters": _parameters,
|
|
83
|
+
"fields": _fields,
|
|
84
|
+
"raw-fields": _raw_fields,
|
|
85
|
+
"calculated-fields": _calculated_fields,
|
|
86
|
+
"joins": _joins,
|
|
87
|
+
"relations": _relations,
|
|
88
|
+
"relationships": _relationships,
|
|
89
|
+
"inferred-relationships": _inferred_relationships,
|
|
90
|
+
"dashboards": _dashboards,
|
|
91
|
+
"dashboard-sheets": _dashboard_sheets,
|
|
92
|
+
"custom-sql": _custom_sql,
|
|
93
|
+
"initial-sql": _initial_sql,
|
|
94
|
+
"published-refs": _published_refs,
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
TABLE_NAMES = list(TABLE_SPECS)
|
twbparser_py/_xml.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Workbook loading and .twbx archive helpers.
|
|
2
|
+
|
|
3
|
+
Ports of `twbx_list`, `extract_twb_from_twbx`, `twbx_extract_files`, and the
|
|
4
|
+
`.twbx_classify` logic from R/utils.R.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import datetime as _dt
|
|
10
|
+
import os
|
|
11
|
+
import tempfile
|
|
12
|
+
import zipfile
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Iterable, Optional
|
|
15
|
+
|
|
16
|
+
import pandas as pd
|
|
17
|
+
from lxml import etree
|
|
18
|
+
|
|
19
|
+
_WORKBOOK_EXT = {"twb"}
|
|
20
|
+
_EXTRACT_EXT = {"hyper", "tde"}
|
|
21
|
+
_IMAGE_EXT = {"png", "jpg", "jpeg", "gif", "svg"}
|
|
22
|
+
_TEXT_EXT = {"csv", "txt", "tsv"}
|
|
23
|
+
_EXCEL_EXT = {"xlsx", "xls"}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _classify(name: str) -> str:
|
|
27
|
+
"""Port of `.twbx_classify`."""
|
|
28
|
+
ext = Path(name).suffix.lower().lstrip(".")
|
|
29
|
+
if ext in _WORKBOOK_EXT:
|
|
30
|
+
return "workbook"
|
|
31
|
+
if ext in _EXTRACT_EXT:
|
|
32
|
+
return "extract"
|
|
33
|
+
if ext in _IMAGE_EXT:
|
|
34
|
+
return "image"
|
|
35
|
+
if ext in _TEXT_EXT:
|
|
36
|
+
return "text"
|
|
37
|
+
if ext in _EXCEL_EXT:
|
|
38
|
+
return "excel"
|
|
39
|
+
return "other"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def twbx_list(twbx_path: str) -> pd.DataFrame:
|
|
43
|
+
"""Port of `twbx_list()`. Lists the contents of a .twbx archive."""
|
|
44
|
+
if not twbx_path or not os.path.exists(twbx_path):
|
|
45
|
+
raise FileNotFoundError(f"File not found: {twbx_path}")
|
|
46
|
+
|
|
47
|
+
rows = []
|
|
48
|
+
with zipfile.ZipFile(twbx_path) as zf:
|
|
49
|
+
for info in zf.infolist():
|
|
50
|
+
if info.is_dir():
|
|
51
|
+
continue
|
|
52
|
+
modified = _dt.datetime(*info.date_time)
|
|
53
|
+
rows.append(
|
|
54
|
+
{
|
|
55
|
+
"name": info.filename,
|
|
56
|
+
"size_bytes": float(info.file_size),
|
|
57
|
+
"modified": modified,
|
|
58
|
+
"type": _classify(info.filename),
|
|
59
|
+
}
|
|
60
|
+
)
|
|
61
|
+
return pd.DataFrame(rows, columns=["name", "size_bytes", "modified", "type"])
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def extract_twb_from_twbx(
|
|
65
|
+
twbx_path: str,
|
|
66
|
+
extract_dir: Optional[str] = None,
|
|
67
|
+
extract_all: bool = False,
|
|
68
|
+
) -> dict:
|
|
69
|
+
"""Port of `extract_twb_from_twbx()`.
|
|
70
|
+
|
|
71
|
+
Extracts the largest `.twb` member (or the whole archive) from a
|
|
72
|
+
`.twbx` file. Returns a dict with `twb_path`, `exdir`, `twbx_path`,
|
|
73
|
+
and `manifest`.
|
|
74
|
+
"""
|
|
75
|
+
if not twbx_path or not os.path.exists(twbx_path):
|
|
76
|
+
raise FileNotFoundError(f"File not found: {twbx_path}")
|
|
77
|
+
|
|
78
|
+
manifest = twbx_list(twbx_path)
|
|
79
|
+
twb_rows = manifest[manifest["type"] == "workbook"].sort_values(
|
|
80
|
+
"size_bytes", ascending=False
|
|
81
|
+
)
|
|
82
|
+
if twb_rows.empty:
|
|
83
|
+
raise ValueError("No .twb file found inside .twbx")
|
|
84
|
+
|
|
85
|
+
twb_rel = twb_rows.iloc[0]["name"]
|
|
86
|
+
|
|
87
|
+
if extract_dir is None:
|
|
88
|
+
stamp = _dt.datetime.now().strftime("%Y%m%d%H%M%S")
|
|
89
|
+
stem = Path(twbx_path).stem
|
|
90
|
+
extract_dir = os.path.join(tempfile.gettempdir(), f"twbx_{stem}_{stamp}")
|
|
91
|
+
|
|
92
|
+
os.makedirs(extract_dir, exist_ok=True)
|
|
93
|
+
with zipfile.ZipFile(twbx_path) as zf:
|
|
94
|
+
if extract_all:
|
|
95
|
+
zf.extractall(extract_dir)
|
|
96
|
+
# extract() sanitizes '..'/absolute segments out of `name` before
|
|
97
|
+
# writing, and returns the *actual* destination path -- computing
|
|
98
|
+
# it ourselves via os.path.join(extract_dir, twb_rel) would
|
|
99
|
+
# silently diverge from where the file really landed for a member
|
|
100
|
+
# name like "../../evil.twb". Calling it again when extract_all
|
|
101
|
+
# already wrote this same member is redundant I/O but harmless
|
|
102
|
+
# (identical bytes), and keeps this one code path authoritative.
|
|
103
|
+
twb_path = zf.extract(twb_rel, extract_dir)
|
|
104
|
+
|
|
105
|
+
return {
|
|
106
|
+
"twb_path": twb_path,
|
|
107
|
+
"exdir": extract_dir,
|
|
108
|
+
"twbx_path": os.path.abspath(twbx_path),
|
|
109
|
+
"manifest": manifest,
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def twbx_extract_files(
|
|
114
|
+
twbx_path: str,
|
|
115
|
+
files: Optional[Iterable[str]] = None,
|
|
116
|
+
pattern: Optional[str] = None,
|
|
117
|
+
types: Optional[Iterable[str]] = None,
|
|
118
|
+
exdir: Optional[str] = None,
|
|
119
|
+
) -> pd.DataFrame:
|
|
120
|
+
"""Port of `twbx_extract_files()`."""
|
|
121
|
+
import re as _re
|
|
122
|
+
|
|
123
|
+
if not twbx_path or not os.path.exists(twbx_path):
|
|
124
|
+
raise FileNotFoundError(f"File not found: {twbx_path}")
|
|
125
|
+
|
|
126
|
+
man = twbx_list(twbx_path)
|
|
127
|
+
sel = man
|
|
128
|
+
if types is not None:
|
|
129
|
+
sel = sel[sel["type"].isin(list(types))]
|
|
130
|
+
if pattern is not None:
|
|
131
|
+
sel = sel[sel["name"].str.contains(pattern, regex=True, na=False)]
|
|
132
|
+
if files is not None:
|
|
133
|
+
sel = sel[sel["name"].isin(list(files))]
|
|
134
|
+
|
|
135
|
+
if sel.empty:
|
|
136
|
+
return pd.DataFrame(columns=["name", "out_path", "type"])
|
|
137
|
+
|
|
138
|
+
if exdir is None:
|
|
139
|
+
stamp = _dt.datetime.now().strftime("%Y%m%d%H%M%S")
|
|
140
|
+
exdir = os.path.join(tempfile.gettempdir(), f"twbx_extract_{stamp}")
|
|
141
|
+
os.makedirs(exdir, exist_ok=True)
|
|
142
|
+
|
|
143
|
+
with zipfile.ZipFile(twbx_path) as zf:
|
|
144
|
+
# extract() returns the sanitized destination path -- see the
|
|
145
|
+
# comment in extract_twb_from_twbx() for why os.path.join(exdir, n)
|
|
146
|
+
# would diverge from it for a member name containing '..'.
|
|
147
|
+
out_paths = [zf.extract(name, exdir) for name in sel["name"]]
|
|
148
|
+
|
|
149
|
+
return pd.DataFrame(
|
|
150
|
+
{
|
|
151
|
+
"name": sel["name"].tolist(),
|
|
152
|
+
"type": sel["type"].tolist(),
|
|
153
|
+
"out_path": out_paths,
|
|
154
|
+
}
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def load_workbook_xml(path: str) -> etree._ElementTree:
|
|
159
|
+
"""Load a .twb/.twbx path into a parsed lxml ElementTree."""
|
|
160
|
+
ext = Path(path).suffix.lower().lstrip(".")
|
|
161
|
+
if ext == "twbx":
|
|
162
|
+
info = extract_twb_from_twbx(path, extract_all=False)
|
|
163
|
+
twb_path = info["twb_path"]
|
|
164
|
+
elif ext == "twb":
|
|
165
|
+
twb_path = path
|
|
166
|
+
else:
|
|
167
|
+
raise ValueError(f"Unsupported file type: {ext}")
|
|
168
|
+
|
|
169
|
+
if not os.path.exists(twb_path):
|
|
170
|
+
raise FileNotFoundError(f"File not found: {twb_path}")
|
|
171
|
+
|
|
172
|
+
return etree.parse(str(twb_path))
|