py-tbparse 0.1.1a0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,168 @@
1
+ Metadata-Version: 2.4
2
+ Name: py-tbparse
3
+ Version: 0.1.1a0
4
+ Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
5
+ Author: DDSNA
6
+ License: MIT
7
+ Keywords: tableau,twb,twbx,workbook,parser,pandas
8
+ Classifier: Development Status :: 4 - Beta
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.9
14
+ Classifier: Programming Language :: Python :: 3.10
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
19
+ Classifier: Topic :: Office/Business
20
+ Requires-Python: >=3.9
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Requires-Dist: lxml>=4.9
24
+ Requires-Dist: pandas>=1.5
25
+ Provides-Extra: test
26
+ Requires-Dist: pytest>=7; extra == "test"
27
+ Provides-Extra: browser
28
+ Requires-Dist: playwright>=1.40; extra == "browser"
29
+ Provides-Extra: dev
30
+ Requires-Dist: build>=1.0; extra == "dev"
31
+ Requires-Dist: twine>=5.0; extra == "dev"
32
+ Dynamic: license-file
33
+
34
+ # twbparser-py
35
+
36
+ A native Python port of the [`twbparser`](https://github.com/PrigasG/twbparser)
37
+ R package: parses Tableau `.twb`/`.twbx` workbook files into `pandas`
38
+ DataFrames. No R runtime required — pure `lxml` XML parsing.
39
+
40
+ This is a v1 subset covering the parser's core: workbook loading,
41
+ datasources, parameters, fields, calculated fields, joins, relationships
42
+ (legacy and 2020.2+), inferred relationships, dashboards, relationship
43
+ validation, custom/initial SQL, and published-source detection.
44
+ Formatting/tooltips/colors/axes/sorts, dashboard layout/actions,
45
+ analytics helpers (calc complexity, field usage, replication brief), and
46
+ the Shiny-inspector equivalent are not yet ported.
47
+
48
+ Beyond the R original, this port also adds a few Python-native extras: a
49
+ Graphviz DOT export of the relationship graph, a workbook-to-workbook
50
+ diff, folder/batch analysis across many workbooks, and Jupyter rich
51
+ display (`_repr_html_`).
52
+
53
+ ## Install
54
+
55
+ ```bash
56
+ pip install -e .
57
+ ```
58
+
59
+ ## Usage
60
+
61
+ ```python
62
+ from twbparser_py import TwbParser
63
+
64
+ p = TwbParser("workbook.twb") # or .twbx
65
+ p.get_datasources()
66
+ p.get_fields()
67
+ p.get_calculated_fields()
68
+ p.get_joins()
69
+ p.get_relationships()
70
+ p.get_inferred_relationships()
71
+ p.get_dashboards()
72
+ p.get_dashboard_sheets()
73
+ p.get_custom_sql()
74
+ p.get_initial_sql()
75
+ p.get_published_refs()
76
+ p.get_relationship_graph_dot() # Graphviz DOT string
77
+ p.validate()
78
+ p.get_overview()
79
+ p # in Jupyter: renders get_overview() via _repr_html_
80
+
81
+ from twbparser_py import diff_workbooks, scan_folder
82
+
83
+ diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
84
+ scan_folder("./workbooks", table="datasources") # one row per workbook x datasource
85
+ ```
86
+
87
+ ### CLI
88
+
89
+ ```bash
90
+ twbparser workbook.twb # overview (default table)
91
+ twbparser workbook.twb tables # list available tables
92
+ twbparser workbook.twb calculated-fields # print a table
93
+ twbparser workbook.twb fields --format csv -o fields.csv
94
+ twbparser workbook.twbx dashboard-sheets --dashboard "Sales Overview"
95
+ twbparser workbook.twb validate # exit code 2 if invalid
96
+ twbparser workbook.twb graph --include-inferred > relationships.dot
97
+ twbparser diff old.twb new.twb datasources # row-level added/removed
98
+ twbparser batch ./workbooks datasources # one table, every workbook in a folder
99
+ ```
100
+
101
+ Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`,
102
+ `calculated-fields`, `joins`, `relations`, `relationships`,
103
+ `inferred-relationships`, `dashboards`, `dashboard-sheets`,
104
+ `custom-sql`, `initial-sql`, `published-refs`. `--format` is `table`
105
+ (default), `csv`, or `json` (`graph` always prints Graphviz DOT text
106
+ regardless of `--format`). `diff`/`batch` accept most of the same table
107
+ names, minus `graph`/`validate`/`tables`.
108
+
109
+ ### GUI
110
+
111
+ A local, browser-based GUI — standard library only (`http.server` +
112
+ vanilla JS), no GUI toolkit or extra dependency required:
113
+
114
+ ```bash
115
+ twbparser-gui workbook.twb # opens your default browser
116
+ twbparser-gui # opens with an empty path field; paste one and click Load
117
+ twbparser-gui --no-browser --port 8765 # just run the server, e.g. for a headless box
118
+ ```
119
+
120
+ Pick a table from the dropdown, filter `dashboard-sheets` by dashboard,
121
+ toggle "include Parameters" for `calculated-fields`, pick `graph` to
122
+ preview/export a Graphviz DOT digraph of the relationships, and export
123
+ any tabular view as CSV. All state lives server-side in memory for the
124
+ life of the process — it's a single-user local tool, not something to
125
+ expose on a shared network.
126
+
127
+ ## Testing
128
+
129
+ ```bash
130
+ pip install -e ".[test]"
131
+ pytest
132
+ ```
133
+
134
+ Fixtures in `tests/fixtures/` are the same tiny sample workbooks used by
135
+ the original R package's test suite (`inst/extdata/`).
136
+
137
+ The GUI additionally has end-to-end tests that drive the page in a real
138
+ headless Chromium (via Playwright) and fail on any uncaught JavaScript
139
+ error. They're opt-in — without the browser installed they skip and the
140
+ rest of the suite runs normally:
141
+
142
+ ```bash
143
+ pip install -e ".[test,browser]"
144
+ playwright install chromium # add --with-deps if you have root
145
+ ./scripts/setup-browser-libs.sh # no-root alternative to --with-deps
146
+ pytest tests/test_gui_browser.py
147
+ ```
148
+
149
+ ## Background
150
+
151
+ `.twb` is plain XML, and `.twbx` is just a zip wrapper around one, which
152
+ is why parsing it natively in Python — no R, no reverse-engineering —
153
+ was tractable at this scale. Power BI's equivalent format, `.pbix`, is a
154
+ binary container built around the proprietary VertiPaq storage engine,
155
+ which is why reading it programmatically needed dedicated
156
+ reverse-engineering projects like
157
+ [PBIXRay](https://github.com/Hugoberry/pbixray) and
158
+ [pbi-tools](https://github.com/pbi-tools/pbi-tools). The comparison
159
+ isn't one-sided, though: Tableau's own official Python tooling for
160
+ *server* automation
161
+ ([`tableauserverclient`](https://pypi.org/project/tableauserverclient/),
162
+ `tabcmd`) is more mature and more open than anything Microsoft ships for
163
+ Power BI's REST API.
164
+
165
+ ## Credit
166
+
167
+ Ported from the R implementation by George Arthur
168
+ ([`PrigasG/twbparser`](https://github.com/PrigasG/twbparser)), MIT licensed.
@@ -0,0 +1,26 @@
1
+ py_tbparse-0.1.1a0.dist-info/licenses/LICENSE,sha256=y8NSwuzSKYNP-QLxH5gd8cYj3m_yNcufgRWRjaC-NCM,1252
2
+ twbparser_py/__init__.py,sha256=Mj5SX7NFlXweWg23nmV9sI6sz34-YtfbAsNlBFAFSps,1856
3
+ twbparser_py/_clean.py,sha256=moCudA2sxmlVI5HBM5EI8BMus_fcMKPOVomeycbOkVY,2313
4
+ twbparser_py/_tables.py,sha256=r-4rXT7Z71XQxn1oz75DC-4wIh2ejTi3UkwgTd2ahwI,2550
5
+ twbparser_py/_xml.py,sha256=C_EjP58cav713GnWrfENga9UZqCVnBRL51H0WcSAcMw,5568
6
+ twbparser_py/batch.py,sha256=WetmMfNYpSGP8cCFCKsP6wYG-lq-wAO-ZS4bCd4APUA,1804
7
+ twbparser_py/calculated_fields.py,sha256=tRo4zCRTJ-4LiMp9QSlq1OTN-Co-snNjLaTALYfvQDo,3859
8
+ twbparser_py/cli.py,sha256=awMN7NnkNX-fQpmnN-Rf3g9KyGQKUD-YLx8t2ztK6I8,7181
9
+ twbparser_py/dashboards.py,sha256=g3fQrqLIc88J6ug3PrCUIXoP8TJV8VZhlG6Av8SM4_U,3282
10
+ twbparser_py/datasources.py,sha256=c924vxYR9_lQKX0GRNQFGuZ3bHXpxQZ7V2Q9PQUoxHA,8790
11
+ twbparser_py/diff.py,sha256=CB1WJUF7cb7H7KOZPObS__SELHmlbL1J3QOb5aHMs1g,3303
12
+ twbparser_py/fields.py,sha256=MlWpdXessuW_mZ8qLZFl6d9-G9coMRz8rxiakEvLKHc,5378
13
+ twbparser_py/graph.py,sha256=sMbh4wayhRxpA7hBNBGiyOF4lGSd_cEKW8sXjdwfw8c,2955
14
+ twbparser_py/joins.py,sha256=Df0O5edFQ9pXjGPgrT8Q4TPOzDP6qNJm0nJ3cGAPewI,3503
15
+ twbparser_py/parser.py,sha256=Ip4hPMCzi8vWeCYCG45kIXt8y6SL0RSngqsYBs4NvQk,8967
16
+ twbparser_py/published.py,sha256=vIFKBI4pw-m9vqXFa5dqY2nGJWsTjN6pAhysj8D2RLI,1472
17
+ twbparser_py/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
18
+ twbparser_py/relationships.py,sha256=uj2k67NEeNG-dCs-uZsVc8ghCtPFeK-Zbfnc8q5KRlk,5669
19
+ twbparser_py/sql.py,sha256=lQonS9GXSAU7pR7yhjCK0WdGBtMTP3bsCXdDkIeWVwI,2443
20
+ twbparser_py/validators.py,sha256=1rG2-1hORa7joNy7cr3PhVfQsASZZjV30PfhItyCSN8,3492
21
+ twbparser_py/webgui.py,sha256=flrdhxf-s0mzOwSKYHIl-r2CZXDYuqtyosUk8-71NdI,14040
22
+ py_tbparse-0.1.1a0.dist-info/METADATA,sha256=OQU06QCv-7AlG9EzCOwVXfWbX1S9yeRW1YqWstv04XU,6402
23
+ py_tbparse-0.1.1a0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
24
+ py_tbparse-0.1.1a0.dist-info/entry_points.txt,sha256=VIH-RDvE5jddcwiDat_CO4nQVLcnTFxfgx4D4vQS39A,93
25
+ py_tbparse-0.1.1a0.dist-info/top_level.txt,sha256=ZEo8_tUfoJEe1tKN5aJCS-mvL_3QAnu8qfELdfaDSnQ,13
26
+ py_tbparse-0.1.1a0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ twbparser = twbparser_py.cli:main
3
+ twbparser-gui = twbparser_py.webgui:main
@@ -0,0 +1,25 @@
1
+ MIT License
2
+
3
+ This project is a Python port of logic originally implemented in the R
4
+ package "twbparser" (https://github.com/PrigasG/twbparser),
5
+ Copyright (c) 2025 George Arthur.
6
+
7
+ Copyright (c) 2026 the twbparser-py contributors
8
+
9
+ Permission is hereby granted, free of charge, to any person obtaining a copy
10
+ of this software and associated documentation files (the "Software"), to deal
11
+ in the Software without restriction, including without limitation the rights
12
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
13
+ copies of the Software, and to permit persons to whom the Software is
14
+ furnished to do so, subject to the following conditions:
15
+
16
+ The above copyright notice and this permission notice shall be included in all
17
+ copies or substantial portions of the Software.
18
+
19
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
20
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
21
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
22
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
23
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
24
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
25
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ twbparser_py
@@ -0,0 +1,54 @@
1
+ """twbparser_py: a native Python port of the twbparser R package.
2
+
3
+ Parses Tableau .twb/.twbx workbook files into pandas DataFrames. Ported
4
+ from https://github.com/PrigasG/twbparser (MIT licensed).
5
+ """
6
+
7
+ from .batch import scan_folder
8
+ from .calculated_fields import extract_calculated_fields, extract_raw_fields
9
+ from .dashboards import dashboard_sheets, list_dashboards
10
+ from .datasources import extract_datasource_details, extract_named_connections, extract_parameters
11
+ from .diff import diff_tables, diff_workbooks
12
+ from .fields import extract_columns_with_table_source, infer_implicit_relationships
13
+ from .graph import to_dot
14
+ from .joins import extract_joins
15
+ from .parser import TwbParser
16
+ from .published import extract_published_refs
17
+ from .relationships import extract_relations, extract_relationships
18
+ from .sql import extract_custom_sql, extract_initial_sql
19
+ from .validators import validate_relationships
20
+ from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
21
+
22
+ __all__ = [
23
+ "TwbParser",
24
+ "extract_calculated_fields",
25
+ "extract_raw_fields",
26
+ "dashboard_sheets",
27
+ "list_dashboards",
28
+ "extract_datasource_details",
29
+ "extract_named_connections",
30
+ "extract_parameters",
31
+ "extract_columns_with_table_source",
32
+ "infer_implicit_relationships",
33
+ "extract_joins",
34
+ "to_dot",
35
+ "extract_relations",
36
+ "extract_relationships",
37
+ "extract_custom_sql",
38
+ "extract_initial_sql",
39
+ "extract_published_refs",
40
+ "validate_relationships",
41
+ "extract_twb_from_twbx",
42
+ "twbx_extract_files",
43
+ "twbx_list",
44
+ "diff_tables",
45
+ "diff_workbooks",
46
+ "scan_folder",
47
+ ]
48
+
49
+ try:
50
+ from importlib.metadata import PackageNotFoundError, version
51
+
52
+ __version__ = version("twbparser-py")
53
+ except PackageNotFoundError: # pragma: no cover - not installed, e.g. running from source
54
+ __version__ = "0.0.0+unknown"
twbparser_py/_clean.py ADDED
@@ -0,0 +1,87 @@
1
+ """Name-cleaning helpers, ported from twbparser's R/utils.R."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from typing import Optional
7
+
8
+ import pandas as pd
9
+
10
+ _TRAILING_HEX32 = re.compile(r"_[0-9A-Fa-f]{32}$")
11
+ _LEADING_BRACKET_PREFIX = re.compile(r"^\[.*?\]\.")
12
+ _BRACKETS = re.compile(r"[\[\]]")
13
+ _DERIVATION_WRAPPER = re.compile(r"^[a-z]+:(.+):[a-z]{1,3}$")
14
+
15
+
16
+ def clean_table(x: Optional[str]) -> Optional[str]:
17
+ """Port of `.twb_clean_table`.
18
+
19
+ Drops `[Extract].`/`[Connection].` prefixes, strips `[]`, and removes
20
+ Tableau's trailing 32-char hex suffix.
21
+ """
22
+ if x is None:
23
+ return None
24
+ s = str(x)
25
+ s = _LEADING_BRACKET_PREFIX.sub("", s)
26
+ s = _BRACKETS.sub("", s)
27
+ s = _TRAILING_HEX32.sub("", s)
28
+ s = s.strip()
29
+ return s if s else None
30
+
31
+
32
+ def clean_field(x: Optional[str]) -> Optional[str]:
33
+ """Port of `.twb_clean_field`.
34
+
35
+ Strips `[]`, takes the last dot-separated token, then unwraps a
36
+ Tableau column-instance wrapper like `none:Category:nk` -> `Category`.
37
+ """
38
+ if x is None:
39
+ return None
40
+ s = str(x)
41
+ s = _BRACKETS.sub("", s)
42
+
43
+ parts = [p for p in s.split(".") if p]
44
+ token = parts[-1] if parts else None
45
+ if token is None:
46
+ return None
47
+
48
+ m = _DERIVATION_WRAPPER.match(token)
49
+ if m:
50
+ return m.group(1)
51
+ return token
52
+
53
+
54
+ def is_missing(x) -> bool:
55
+ """True for any "missing" scalar (`None`, float `NaN`, `NaT`, `pd.NA`,
56
+ ...) -- not a port of an R function, but a shared helper so callers
57
+ (`diff.py`, `validators.py`) don't each reimplement `pd.isna()`'s
58
+ edge cases (e.g. it raises on array-likes) slightly differently."""
59
+ if x is None:
60
+ return True
61
+ try:
62
+ return bool(pd.isna(x))
63
+ except (TypeError, ValueError):
64
+ return False
65
+
66
+
67
+ def attr_safe_get(attrs: dict, name: str, default=None):
68
+ """Port of `attr_safe_get`."""
69
+ if attrs is None:
70
+ return default
71
+ return attrs.get(name, default)
72
+
73
+
74
+ def strip_brackets(x: Optional[str]) -> Optional[str]:
75
+ """Port of `.strip_brackets`."""
76
+ if x is None:
77
+ return None
78
+ return _BRACKETS.sub("", x)
79
+
80
+
81
+ def basename_safe(x: Optional[str], fallback: str = "<unknown>") -> str:
82
+ """Port of `basename_safe`."""
83
+ if not x:
84
+ return fallback
85
+ import os
86
+
87
+ return os.path.basename(x)
@@ -0,0 +1,97 @@
1
+ """Shared table registry used by both the CLI and the web GUI.
2
+
3
+ Keeping this in one place means the CLI and GUI can never drift out of
4
+ sync on which tables exist or how their optional parameters
5
+ (`dashboard`, `include_parameters`) are threaded through to `TwbParser`.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import Callable
11
+
12
+ import pandas as pd
13
+
14
+ from .parser import TwbParser
15
+
16
+
17
+ def _overview(p: TwbParser, **_kw) -> pd.DataFrame:
18
+ return p.get_overview()
19
+
20
+
21
+ def _datasources(p: TwbParser, **_kw) -> pd.DataFrame:
22
+ return p.get_datasources()
23
+
24
+
25
+ def _parameters(p: TwbParser, **_kw) -> pd.DataFrame:
26
+ return p.get_parameters()
27
+
28
+
29
+ def _fields(p: TwbParser, **_kw) -> pd.DataFrame:
30
+ return p.get_fields()
31
+
32
+
33
+ def _raw_fields(p: TwbParser, **_kw) -> pd.DataFrame:
34
+ return p.get_raw_fields()
35
+
36
+
37
+ def _calculated_fields(p: TwbParser, include_parameters: bool = False, **_kw) -> pd.DataFrame:
38
+ return p.get_calculated_fields(include_parameters=include_parameters)
39
+
40
+
41
+ def _joins(p: TwbParser, **_kw) -> pd.DataFrame:
42
+ return p.get_joins()
43
+
44
+
45
+ def _relations(p: TwbParser, **_kw) -> pd.DataFrame:
46
+ return p.get_relations()
47
+
48
+
49
+ def _relationships(p: TwbParser, **_kw) -> pd.DataFrame:
50
+ return p.get_relationships()
51
+
52
+
53
+ def _inferred_relationships(p: TwbParser, **_kw) -> pd.DataFrame:
54
+ return p.get_inferred_relationships()
55
+
56
+
57
+ def _dashboards(p: TwbParser, **_kw) -> pd.DataFrame:
58
+ return p.get_dashboards()
59
+
60
+
61
+ def _dashboard_sheets(p: TwbParser, dashboard: str | None = None, **_kw) -> pd.DataFrame:
62
+ return p.get_dashboard_sheets(dashboard=dashboard or None)
63
+
64
+
65
+ def _custom_sql(p: TwbParser, **_kw) -> pd.DataFrame:
66
+ return p.get_custom_sql()
67
+
68
+
69
+ def _initial_sql(p: TwbParser, **_kw) -> pd.DataFrame:
70
+ return p.get_initial_sql()
71
+
72
+
73
+ def _published_refs(p: TwbParser, **_kw) -> pd.DataFrame:
74
+ return p.get_published_refs()
75
+
76
+
77
+ # Order here is display order in both the CLI's `tables` listing and the
78
+ # GUI's table dropdown.
79
+ TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
80
+ "overview": _overview,
81
+ "datasources": _datasources,
82
+ "parameters": _parameters,
83
+ "fields": _fields,
84
+ "raw-fields": _raw_fields,
85
+ "calculated-fields": _calculated_fields,
86
+ "joins": _joins,
87
+ "relations": _relations,
88
+ "relationships": _relationships,
89
+ "inferred-relationships": _inferred_relationships,
90
+ "dashboards": _dashboards,
91
+ "dashboard-sheets": _dashboard_sheets,
92
+ "custom-sql": _custom_sql,
93
+ "initial-sql": _initial_sql,
94
+ "published-refs": _published_refs,
95
+ }
96
+
97
+ TABLE_NAMES = list(TABLE_SPECS)
twbparser_py/_xml.py ADDED
@@ -0,0 +1,172 @@
1
+ """Workbook loading and .twbx archive helpers.
2
+
3
+ Ports of `twbx_list`, `extract_twb_from_twbx`, `twbx_extract_files`, and the
4
+ `.twbx_classify` logic from R/utils.R.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import datetime as _dt
10
+ import os
11
+ import tempfile
12
+ import zipfile
13
+ from pathlib import Path
14
+ from typing import Iterable, Optional
15
+
16
+ import pandas as pd
17
+ from lxml import etree
18
+
19
+ _WORKBOOK_EXT = {"twb"}
20
+ _EXTRACT_EXT = {"hyper", "tde"}
21
+ _IMAGE_EXT = {"png", "jpg", "jpeg", "gif", "svg"}
22
+ _TEXT_EXT = {"csv", "txt", "tsv"}
23
+ _EXCEL_EXT = {"xlsx", "xls"}
24
+
25
+
26
+ def _classify(name: str) -> str:
27
+ """Port of `.twbx_classify`."""
28
+ ext = Path(name).suffix.lower().lstrip(".")
29
+ if ext in _WORKBOOK_EXT:
30
+ return "workbook"
31
+ if ext in _EXTRACT_EXT:
32
+ return "extract"
33
+ if ext in _IMAGE_EXT:
34
+ return "image"
35
+ if ext in _TEXT_EXT:
36
+ return "text"
37
+ if ext in _EXCEL_EXT:
38
+ return "excel"
39
+ return "other"
40
+
41
+
42
+ def twbx_list(twbx_path: str) -> pd.DataFrame:
43
+ """Port of `twbx_list()`. Lists the contents of a .twbx archive."""
44
+ if not twbx_path or not os.path.exists(twbx_path):
45
+ raise FileNotFoundError(f"File not found: {twbx_path}")
46
+
47
+ rows = []
48
+ with zipfile.ZipFile(twbx_path) as zf:
49
+ for info in zf.infolist():
50
+ if info.is_dir():
51
+ continue
52
+ modified = _dt.datetime(*info.date_time)
53
+ rows.append(
54
+ {
55
+ "name": info.filename,
56
+ "size_bytes": float(info.file_size),
57
+ "modified": modified,
58
+ "type": _classify(info.filename),
59
+ }
60
+ )
61
+ return pd.DataFrame(rows, columns=["name", "size_bytes", "modified", "type"])
62
+
63
+
64
+ def extract_twb_from_twbx(
65
+ twbx_path: str,
66
+ extract_dir: Optional[str] = None,
67
+ extract_all: bool = False,
68
+ ) -> dict:
69
+ """Port of `extract_twb_from_twbx()`.
70
+
71
+ Extracts the largest `.twb` member (or the whole archive) from a
72
+ `.twbx` file. Returns a dict with `twb_path`, `exdir`, `twbx_path`,
73
+ and `manifest`.
74
+ """
75
+ if not twbx_path or not os.path.exists(twbx_path):
76
+ raise FileNotFoundError(f"File not found: {twbx_path}")
77
+
78
+ manifest = twbx_list(twbx_path)
79
+ twb_rows = manifest[manifest["type"] == "workbook"].sort_values(
80
+ "size_bytes", ascending=False
81
+ )
82
+ if twb_rows.empty:
83
+ raise ValueError("No .twb file found inside .twbx")
84
+
85
+ twb_rel = twb_rows.iloc[0]["name"]
86
+
87
+ if extract_dir is None:
88
+ stamp = _dt.datetime.now().strftime("%Y%m%d%H%M%S")
89
+ stem = Path(twbx_path).stem
90
+ extract_dir = os.path.join(tempfile.gettempdir(), f"twbx_{stem}_{stamp}")
91
+
92
+ os.makedirs(extract_dir, exist_ok=True)
93
+ with zipfile.ZipFile(twbx_path) as zf:
94
+ if extract_all:
95
+ zf.extractall(extract_dir)
96
+ # extract() sanitizes '..'/absolute segments out of `name` before
97
+ # writing, and returns the *actual* destination path -- computing
98
+ # it ourselves via os.path.join(extract_dir, twb_rel) would
99
+ # silently diverge from where the file really landed for a member
100
+ # name like "../../evil.twb". Calling it again when extract_all
101
+ # already wrote this same member is redundant I/O but harmless
102
+ # (identical bytes), and keeps this one code path authoritative.
103
+ twb_path = zf.extract(twb_rel, extract_dir)
104
+
105
+ return {
106
+ "twb_path": twb_path,
107
+ "exdir": extract_dir,
108
+ "twbx_path": os.path.abspath(twbx_path),
109
+ "manifest": manifest,
110
+ }
111
+
112
+
113
+ def twbx_extract_files(
114
+ twbx_path: str,
115
+ files: Optional[Iterable[str]] = None,
116
+ pattern: Optional[str] = None,
117
+ types: Optional[Iterable[str]] = None,
118
+ exdir: Optional[str] = None,
119
+ ) -> pd.DataFrame:
120
+ """Port of `twbx_extract_files()`."""
121
+ import re as _re
122
+
123
+ if not twbx_path or not os.path.exists(twbx_path):
124
+ raise FileNotFoundError(f"File not found: {twbx_path}")
125
+
126
+ man = twbx_list(twbx_path)
127
+ sel = man
128
+ if types is not None:
129
+ sel = sel[sel["type"].isin(list(types))]
130
+ if pattern is not None:
131
+ sel = sel[sel["name"].str.contains(pattern, regex=True, na=False)]
132
+ if files is not None:
133
+ sel = sel[sel["name"].isin(list(files))]
134
+
135
+ if sel.empty:
136
+ return pd.DataFrame(columns=["name", "out_path", "type"])
137
+
138
+ if exdir is None:
139
+ stamp = _dt.datetime.now().strftime("%Y%m%d%H%M%S")
140
+ exdir = os.path.join(tempfile.gettempdir(), f"twbx_extract_{stamp}")
141
+ os.makedirs(exdir, exist_ok=True)
142
+
143
+ with zipfile.ZipFile(twbx_path) as zf:
144
+ # extract() returns the sanitized destination path -- see the
145
+ # comment in extract_twb_from_twbx() for why os.path.join(exdir, n)
146
+ # would diverge from it for a member name containing '..'.
147
+ out_paths = [zf.extract(name, exdir) for name in sel["name"]]
148
+
149
+ return pd.DataFrame(
150
+ {
151
+ "name": sel["name"].tolist(),
152
+ "type": sel["type"].tolist(),
153
+ "out_path": out_paths,
154
+ }
155
+ )
156
+
157
+
158
+ def load_workbook_xml(path: str) -> etree._ElementTree:
159
+ """Load a .twb/.twbx path into a parsed lxml ElementTree."""
160
+ ext = Path(path).suffix.lower().lstrip(".")
161
+ if ext == "twbx":
162
+ info = extract_twb_from_twbx(path, extract_all=False)
163
+ twb_path = info["twb_path"]
164
+ elif ext == "twb":
165
+ twb_path = path
166
+ else:
167
+ raise ValueError(f"Unsupported file type: {ext}")
168
+
169
+ if not os.path.exists(twb_path):
170
+ raise FileNotFoundError(f"File not found: {twb_path}")
171
+
172
+ return etree.parse(str(twb_path))