py-tbparse 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/AGENTS.md +10 -9
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/LICENSE +1 -1
- py_tbparse-0.3.0/PKG-INFO +196 -0
- py_tbparse-0.3.0/README.md +160 -0
- py_tbparse-0.3.0/docs/gui-field-renames.png +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/__init__.py +13 -1
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/_tables.py +8 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/cli.py +114 -8
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/parser.py +14 -0
- py_tbparse-0.3.0/py_tbparse/rename.py +520 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/webgui.py +198 -15
- py_tbparse-0.3.0/py_tbparse.egg-info/PKG-INFO +196 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/py_tbparse.egg-info/SOURCES.txt +26 -21
- py_tbparse-0.3.0/py_tbparse.egg-info/entry_points.txt +3 -0
- py_tbparse-0.3.0/py_tbparse.egg-info/top_level.txt +1 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/pyproject.toml +5 -5
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_batch.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_calculated_fields.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_clean.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_cli.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_dashboards.py +2 -2
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_datasources.py +2 -2
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_diff.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_fields.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_graph.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_gui_browser.py +49 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_joins.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_parser.py +2 -2
- py_tbparse-0.3.0/tests/test_public_workbooks.py +31 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_published.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_relationships.py +3 -3
- py_tbparse-0.3.0/tests/test_rename.py +386 -0
- py_tbparse-0.3.0/tests/test_rename_mapping.py +68 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_sql.py +1 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_validators.py +2 -2
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_version.py +3 -3
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_webgui.py +80 -2
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_xml.py +2 -2
- py_tbparse-0.2.0/PKG-INFO +0 -150
- py_tbparse-0.2.0/README.md +0 -114
- py_tbparse-0.2.0/py_tbparse.egg-info/PKG-INFO +0 -150
- py_tbparse-0.2.0/py_tbparse.egg-info/entry_points.txt +0 -3
- py_tbparse-0.2.0/py_tbparse.egg-info/top_level.txt +0 -1
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/MANIFEST.in +0 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/docs/gui-overview.png +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/_clean.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/_xml.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/batch.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/calculated_fields.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/dashboards.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/datasources.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/diff.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/fields.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/graph.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/joins.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/published.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/py.typed +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/relationships.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/sql.py +0 -0
- {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/validators.py +0 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/py_tbparse.egg-info/dependency_links.txt +0 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/py_tbparse.egg-info/requires.txt +0 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/setup.cfg +0 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/conftest.py +0 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/fixtures/test_for_wenjie.twb +0 -0
- {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/fixtures/test_for_zip.twbx +0 -0
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What this is
|
|
4
4
|
|
|
5
|
-
`
|
|
5
|
+
`py_tbparse` is a native Python port of the R package
|
|
6
6
|
[`twbparser`](https://github.com/PrigasG/twbparser) (mirrored at
|
|
7
7
|
`DDSNA/twbparser`): it parses Tableau `.twb`/`.twbx` workbook files into
|
|
8
8
|
`pandas` DataFrames. Pure `lxml` XML parsing — no R runtime, no `rpy2`.
|
|
@@ -37,7 +37,7 @@ python3 -m venv .venv
|
|
|
37
37
|
```bash
|
|
38
38
|
.venv/bin/python -m pytest -q # run the full test suite
|
|
39
39
|
.venv/bin/python -m pytest -q tests/test_joins.py # single file
|
|
40
|
-
.venv/bin/python -m py_compile
|
|
40
|
+
.venv/bin/python -m py_compile py_tbparse/*.py # syntax check
|
|
41
41
|
```
|
|
42
42
|
|
|
43
43
|
Browser (GUI) tests need a one-time setup; without it they skip and the
|
|
@@ -58,19 +58,19 @@ via `from __future__ import annotations`).
|
|
|
58
58
|
|
|
59
59
|
```bash
|
|
60
60
|
.venv/bin/pip install -e ".[dev]" # build + twine
|
|
61
|
-
rm -rf dist build
|
|
61
|
+
rm -rf dist build py_tbparse.egg-info
|
|
62
62
|
.venv/bin/python -m build # produces dist/*.whl and dist/*.tar.gz
|
|
63
63
|
.venv/bin/twine check dist/* # validates metadata/README rendering
|
|
64
64
|
```
|
|
65
65
|
|
|
66
66
|
Version is single-sourced from `pyproject.toml`'s `[project].version`;
|
|
67
|
-
`
|
|
67
|
+
`py_tbparse.__version__` reads it back via `importlib.metadata` at
|
|
68
68
|
runtime (see `__init__.py`), so don't hardcode a second copy.
|
|
69
69
|
|
|
70
70
|
Before bumping the version for a release: bump `version` in
|
|
71
71
|
`pyproject.toml`, then rebuild and smoke-test the wheel in a
|
|
72
|
-
throwaway venv (`pip install dist/*.whl`, run `
|
|
73
|
-
`
|
|
72
|
+
throwaway venv (`pip install dist/*.whl`, run `py-tbparse --help` and
|
|
73
|
+
`py-tbparse-gui --help`, run pytest against an extracted sdist) — this
|
|
74
74
|
catches packaging bugs (missing files, wrong entry points) that an
|
|
75
75
|
editable install won't.
|
|
76
76
|
|
|
@@ -88,7 +88,7 @@ Publishers" settings page before the first release, and needs a
|
|
|
88
88
|
|
|
89
89
|
## Architecture
|
|
90
90
|
|
|
91
|
-
Each `
|
|
91
|
+
Each `py_tbparse/*.py` module is a direct port of one R source file in
|
|
92
92
|
the upstream package, function-for-function:
|
|
93
93
|
|
|
94
94
|
| Python module | Ported from (R) | Notes |
|
|
@@ -120,10 +120,11 @@ beyond the R package's scope:
|
|
|
120
120
|
| Python module | What it is |
|
|
121
121
|
|---|---|
|
|
122
122
|
| `_tables.py` | Name → `TwbParser`-accessor registry shared by `cli.py`, `webgui.py`, `diff.py`, and `batch.py`. Adding a new extractor to `TwbParser`? Add it here too so it's automatically available everywhere else. |
|
|
123
|
-
| `cli.py` | `
|
|
124
|
-
| `webgui.py` | `
|
|
123
|
+
| `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
|
|
124
|
+
| `webgui.py` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
|
|
125
125
|
| `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
|
|
126
126
|
| `diff.py` | `diff_tables()`/`diff_workbooks()`: row-level added/removed diff between two workbooks' same-named table, via `_tables.TABLE_SPECS`. No "changed" classification without a natural key — a changed row shows as one removed + one added row. |
|
|
127
|
+
| `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` table in `_tables.py`. |
|
|
127
128
|
| `batch.py` | `scan_folder()`: runs one table across every `.twb`/`.twbx` in a directory, concatenated with a `workbook` column. Skips (with a warning) any file that fails to load/extract rather than aborting the batch. |
|
|
128
129
|
|
|
129
130
|
## Porting conventions (read before adding/modifying a function)
|
|
@@ -4,7 +4,7 @@ This project is a Python port of logic originally implemented in the R
|
|
|
4
4
|
package "twbparser" (https://github.com/PrigasG/twbparser),
|
|
5
5
|
Copyright (c) 2025 George Arthur.
|
|
6
6
|
|
|
7
|
-
Copyright (c) 2026 the
|
|
7
|
+
Copyright (c) 2026 the py-tbparse contributors
|
|
8
8
|
|
|
9
9
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
10
10
|
of this software and associated documentation files (the "Software"), to deal
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: py-tbparse
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
|
|
5
|
+
Author: DDSNA
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/DDSNA/py-tbparse
|
|
8
|
+
Project-URL: Source, https://github.com/DDSNA/py-tbparse
|
|
9
|
+
Project-URL: Issues, https://github.com/DDSNA/py-tbparse/issues
|
|
10
|
+
Keywords: tableau,twb,twbx,workbook,parser,pandas
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
+
Classifier: Topic :: Office/Business
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: lxml>=4.9
|
|
27
|
+
Requires-Dist: pandas>=1.5
|
|
28
|
+
Provides-Extra: test
|
|
29
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
30
|
+
Provides-Extra: browser
|
|
31
|
+
Requires-Dist: playwright>=1.40; extra == "browser"
|
|
32
|
+
Provides-Extra: dev
|
|
33
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
34
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# py-tbparse
|
|
38
|
+
|
|
39
|
+
Reads Tableau workbooks (`.twb` and `.twbx`) and gives you what's in them as pandas DataFrames: datasources, fields, calculated fields, joins, relationships, dashboards, custom SQL. It's plain Python. You don't need Tableau or R installed.
|
|
40
|
+
|
|
41
|
+
It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The browser GUI, the command-line tool, workbook diffing and folder scanning are new here.
|
|
42
|
+
|
|
43
|
+

|
|
44
|
+
|
|
45
|
+
## Install
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install py-tbparse
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Or from a checkout:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
git clone https://github.com/DDSNA/py-tbparse.git
|
|
55
|
+
cd py-tbparse
|
|
56
|
+
pip install -e .
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
One name throughout: `pip install py-tbparse`, `import py_tbparse`, run `py-tbparse` or `py-tbparse-gui`. (Releases up to 0.2.0 imported `twbparser_py` and ran `twbparser` / `twbparser-gui`; those names are gone.)
|
|
60
|
+
|
|
61
|
+
## Using it from Python
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from py_tbparse import TwbParser
|
|
65
|
+
|
|
66
|
+
p = TwbParser("workbook.twb") # or a .twbx
|
|
67
|
+
p.get_overview() # counts of everything
|
|
68
|
+
p.get_calculated_fields() # each calculation and its formula
|
|
69
|
+
p.get_relationships() # how the tables connect
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Every getter returns a DataFrame. The rest are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql` and `get_published_refs`. `get_relationship_graph_dot()` returns the data model as Graphviz text, and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
|
|
73
|
+
|
|
74
|
+
Two helpers work across workbooks:
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
from py_tbparse import diff_workbooks, scan_folder
|
|
78
|
+
|
|
79
|
+
diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
|
|
80
|
+
scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
### Cleaning field names after a datasource switch
|
|
84
|
+
|
|
85
|
+
Pointing a workbook at a new datasource that only partly matches the old schema tends to leave ugly names: `ORDER_ID`, `orderId`, `Order ID (Orders1)`, `Order ID1`. `suggest_field_renames` proposes a clean name for each field. It only reports; it never edits the workbook.
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
from py_tbparse import TwbParser, suggest_field_renames
|
|
89
|
+
|
|
90
|
+
new = TwbParser("after_switch.twb")
|
|
91
|
+
old = TwbParser("before_switch.twb") # optional: the schema you want to match
|
|
92
|
+
|
|
93
|
+
suggest_field_renames(new, reference=old, only_changed=True)
|
|
94
|
+
# name current suggested reason score
|
|
95
|
+
# [ORDER_ID] ORDER_ID Order ID matches reference 1.0
|
|
96
|
+
# [Sales Amount (Orders1)] Sales Amount (Orders1) Sales Amount matches reference 1.0
|
|
97
|
+
# [orderDate] orderDate Order Date normalized NaN
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
With a `reference` (a workbook, a fields table or a plain list of names), fields that match by name ignoring case, separators and Tableau's duplicate suffixes take the reference's exact spelling, and near-misses above `fuzzy_cutoff` (default 0.85) are matched too. Everything else is tidied by `normalize_name(name, style)`, where `style` is `title` (default), `snake`, `lower` or `keep`. Names that are already clean (`YTD Sales`, `iPhone Units`, `Country/Region`) are left alone. `1` and `(Table1)` suffixes are only dropped when the plain name exists in the same datasource (`Address Line 2` and `Q1` are never treated as duplicates), fuzzy matches never cross a different number, and two fields never get the same suggestion (the loser stays as it is, with `reason` set to `conflict`).
|
|
101
|
+
|
|
102
|
+
`p.get_field_renames()` does the same from a parser.
|
|
103
|
+
|
|
104
|
+
**Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
py-tbparse rename new.twb -r old.twb -f csv -o mapping.csv
|
|
108
|
+
py-tbparse rename new.twb --apply mapping.csv # writes new_renamed.twb; add --write-workbook PATH to choose the file
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Your edits are applied as written, including rows the tool had marked `conflict`; two rows giving the same name in one datasource are rejected. From Python this is `apply_field_renames(p, renames=load_rename_mapping("mapping.csv"))`.
|
|
112
|
+
|
|
113
|
+
**What will stay broken.** `py-tbparse rename new.twb -r old.twb --missing` (or `compare_field_schemas(new, old)`) lists the fields with no counterpart after the switch: `old only` fields that nothing in the new source matches, and `new only` fields nothing in the old workbook matches, each with the closest name on the other side as a hint. Sheets using an `old only` field stay red after Replace Data Source until you map or recreate it.
|
|
114
|
+
|
|
115
|
+
To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...)`) writes `<name>_renamed.twb` / `.twbx` next to the original. It sets each field's caption, which is how Tableau renames a field; the internal names that formulas and sheets use are not touched, and a `.twbx` keeps all its other contents. A field that only exists as a physical column (typical right after a datasource switch) gets a new minimal `<column>` element carrying the caption; that shape follows what Tableau writes but I have not opened such files in Tableau itself. It never modifies the original and refuses to overwrite an existing file unless you pass `overwrite=True`.
|
|
116
|
+
|
|
117
|
+
**When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
|
|
118
|
+
|
|
119
|
+
### `.twbx` files
|
|
120
|
+
|
|
121
|
+
A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
from py_tbparse import extract_twb_from_twbx, twbx_extract_files
|
|
125
|
+
|
|
126
|
+
extract_twb_from_twbx("workbook.twbx", extract_dir="out/")
|
|
127
|
+
twbx_extract_files("workbook.twbx", exdir="out/") # everything in the package
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
## Command line
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
py-tbparse workbook.twb # overview
|
|
134
|
+
py-tbparse workbook.twb tables # list the tables
|
|
135
|
+
py-tbparse workbook.twb calculated-fields
|
|
136
|
+
py-tbparse workbook.twb fields --format csv -o fields.csv
|
|
137
|
+
py-tbparse workbook.twbx dashboard-sheets --dashboard "Sales Overview"
|
|
138
|
+
py-tbparse workbook.twb validate # exit code 2 if it finds a problem
|
|
139
|
+
py-tbparse workbook.twb graph > model.dot # --include-inferred adds the guessed links
|
|
140
|
+
py-tbparse diff old.twb new.twb datasources
|
|
141
|
+
py-tbparse batch ./workbooks datasources
|
|
142
|
+
py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
|
|
143
|
+
py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`.
|
|
147
|
+
|
|
148
|
+
`--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
|
|
149
|
+
|
|
150
|
+
## GUI
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
py-tbparse-gui workbook.twb # opens your browser with it loaded
|
|
154
|
+
py-tbparse-gui # starts empty, paste a path and hit Load
|
|
155
|
+
py-tbparse-gui --no-browser --port 8765 # server only, for a machine with no display
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
It uses only the standard library, so there's nothing more to install.
|
|
159
|
+
|
|
160
|
+
The sidebar lists every table with its row count. Click a column header to sort, type in the filter box to narrow rows (`/` jumps to it), click a row to see a long formula or SQL statement in full. The tiles on the overview open their tables. The graph view lets you copy or download the DOT text. The Field renames view has the buttons for the feature above: pick a style, datasource and optional reference workbook, then **Create fixed workbook** saves `<name>_renamed` beside the original (and says so if that file already exists) or **Download fixed workbook** sends it to your browser without saving anything. Everything else exports as CSV. It has a dark theme and works in a narrow window.
|
|
161
|
+
|
|
162
|
+

|
|
163
|
+
|
|
164
|
+
It's meant to run on your own machine for one person. It refuses requests that come from other websites, but there's no login, so don't put it on a shared network.
|
|
165
|
+
|
|
166
|
+
## Tests
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
pip install -e ".[test]"
|
|
170
|
+
pytest
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
|
|
174
|
+
|
|
175
|
+
The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
pip install -e ".[test,browser]"
|
|
179
|
+
playwright install chromium # add --with-deps if you have root
|
|
180
|
+
./scripts/setup-browser-libs.sh # without root, this unpacks the system libraries locally
|
|
181
|
+
pytest tests/test_gui_browser.py
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
## Compared with the R package
|
|
185
|
+
|
|
186
|
+
The code follows the R original function by function, and the aim is the same output. Not everything is ported. Missing so far: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, the analytics helpers (calculation complexity, field usage, replication brief) and the Shiny inspector. The GUI covers some of what the inspector did.
|
|
187
|
+
|
|
188
|
+
Where the R version has a bug, this one doesn't copy it. That currently covers joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
|
|
189
|
+
|
|
190
|
+
## Why a Tableau parser
|
|
191
|
+
|
|
192
|
+
A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python isn't hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
|
|
193
|
+
|
|
194
|
+
## Credit
|
|
195
|
+
|
|
196
|
+
Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
# py-tbparse
|
|
2
|
+
|
|
3
|
+
Reads Tableau workbooks (`.twb` and `.twbx`) and gives you what's in them as pandas DataFrames: datasources, fields, calculated fields, joins, relationships, dashboards, custom SQL. It's plain Python. You don't need Tableau or R installed.
|
|
4
|
+
|
|
5
|
+
It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The browser GUI, the command-line tool, workbook diffing and folder scanning are new here.
|
|
6
|
+
|
|
7
|
+

|
|
8
|
+
|
|
9
|
+
## Install
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install py-tbparse
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Or from a checkout:
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
git clone https://github.com/DDSNA/py-tbparse.git
|
|
19
|
+
cd py-tbparse
|
|
20
|
+
pip install -e .
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
One name throughout: `pip install py-tbparse`, `import py_tbparse`, run `py-tbparse` or `py-tbparse-gui`. (Releases up to 0.2.0 imported `twbparser_py` and ran `twbparser` / `twbparser-gui`; those names are gone.)
|
|
24
|
+
|
|
25
|
+
## Using it from Python
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
from py_tbparse import TwbParser
|
|
29
|
+
|
|
30
|
+
p = TwbParser("workbook.twb") # or a .twbx
|
|
31
|
+
p.get_overview() # counts of everything
|
|
32
|
+
p.get_calculated_fields() # each calculation and its formula
|
|
33
|
+
p.get_relationships() # how the tables connect
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Every getter returns a DataFrame. The rest are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql` and `get_published_refs`. `get_relationship_graph_dot()` returns the data model as Graphviz text, and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
|
|
37
|
+
|
|
38
|
+
Two helpers work across workbooks:
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from py_tbparse import diff_workbooks, scan_folder
|
|
42
|
+
|
|
43
|
+
diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
|
|
44
|
+
scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
### Cleaning field names after a datasource switch
|
|
48
|
+
|
|
49
|
+
Pointing a workbook at a new datasource that only partly matches the old schema tends to leave ugly names: `ORDER_ID`, `orderId`, `Order ID (Orders1)`, `Order ID1`. `suggest_field_renames` proposes a clean name for each field. It only reports; it never edits the workbook.
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from py_tbparse import TwbParser, suggest_field_renames
|
|
53
|
+
|
|
54
|
+
new = TwbParser("after_switch.twb")
|
|
55
|
+
old = TwbParser("before_switch.twb") # optional: the schema you want to match
|
|
56
|
+
|
|
57
|
+
suggest_field_renames(new, reference=old, only_changed=True)
|
|
58
|
+
# name current suggested reason score
|
|
59
|
+
# [ORDER_ID] ORDER_ID Order ID matches reference 1.0
|
|
60
|
+
# [Sales Amount (Orders1)] Sales Amount (Orders1) Sales Amount matches reference 1.0
|
|
61
|
+
# [orderDate] orderDate Order Date normalized NaN
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
With a `reference` (a workbook, a fields table or a plain list of names), fields that match by name ignoring case, separators and Tableau's duplicate suffixes take the reference's exact spelling, and near-misses above `fuzzy_cutoff` (default 0.85) are matched too. Everything else is tidied by `normalize_name(name, style)`, where `style` is `title` (default), `snake`, `lower` or `keep`. Names that are already clean (`YTD Sales`, `iPhone Units`, `Country/Region`) are left alone. `1` and `(Table1)` suffixes are only dropped when the plain name exists in the same datasource (`Address Line 2` and `Q1` are never treated as duplicates), fuzzy matches never cross a different number, and two fields never get the same suggestion (the loser stays as it is, with `reason` set to `conflict`).
|
|
65
|
+
|
|
66
|
+
`p.get_field_renames()` does the same from a parser.
|
|
67
|
+
|
|
68
|
+
**Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
py-tbparse rename new.twb -r old.twb -f csv -o mapping.csv
|
|
72
|
+
py-tbparse rename new.twb --apply mapping.csv # writes new_renamed.twb; add --write-workbook PATH to choose the file
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Your edits are applied as written, including rows the tool had marked `conflict`; two rows giving the same name in one datasource are rejected. From Python this is `apply_field_renames(p, renames=load_rename_mapping("mapping.csv"))`.
|
|
76
|
+
|
|
77
|
+
**What will stay broken.** `py-tbparse rename new.twb -r old.twb --missing` (or `compare_field_schemas(new, old)`) lists the fields with no counterpart after the switch: `old only` fields that nothing in the new source matches, and `new only` fields nothing in the old workbook matches, each with the closest name on the other side as a hint. Sheets using an `old only` field stay red after Replace Data Source until you map or recreate it.
|
|
78
|
+
|
|
79
|
+
To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...)`) writes `<name>_renamed.twb` / `.twbx` next to the original. It sets each field's caption, which is how Tableau renames a field; the internal names that formulas and sheets use are not touched, and a `.twbx` keeps all its other contents. A field that only exists as a physical column (typical right after a datasource switch) gets a new minimal `<column>` element carrying the caption; that shape follows what Tableau writes but I have not opened such files in Tableau itself. It never modifies the original and refuses to overwrite an existing file unless you pass `overwrite=True`.
|
|
80
|
+
|
|
81
|
+
**When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
|
|
82
|
+
|
|
83
|
+
### `.twbx` files
|
|
84
|
+
|
|
85
|
+
A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
from py_tbparse import extract_twb_from_twbx, twbx_extract_files
|
|
89
|
+
|
|
90
|
+
extract_twb_from_twbx("workbook.twbx", extract_dir="out/")
|
|
91
|
+
twbx_extract_files("workbook.twbx", exdir="out/") # everything in the package
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Command line
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
py-tbparse workbook.twb # overview
|
|
98
|
+
py-tbparse workbook.twb tables # list the tables
|
|
99
|
+
py-tbparse workbook.twb calculated-fields
|
|
100
|
+
py-tbparse workbook.twb fields --format csv -o fields.csv
|
|
101
|
+
py-tbparse workbook.twbx dashboard-sheets --dashboard "Sales Overview"
|
|
102
|
+
py-tbparse workbook.twb validate # exit code 2 if it finds a problem
|
|
103
|
+
py-tbparse workbook.twb graph > model.dot # --include-inferred adds the guessed links
|
|
104
|
+
py-tbparse diff old.twb new.twb datasources
|
|
105
|
+
py-tbparse batch ./workbooks datasources
|
|
106
|
+
py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
|
|
107
|
+
py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`.
|
|
111
|
+
|
|
112
|
+
`--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
|
|
113
|
+
|
|
114
|
+
## GUI
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
py-tbparse-gui workbook.twb # opens your browser with it loaded
|
|
118
|
+
py-tbparse-gui # starts empty, paste a path and hit Load
|
|
119
|
+
py-tbparse-gui --no-browser --port 8765 # server only, for a machine with no display
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
It uses only the standard library, so there's nothing more to install.
|
|
123
|
+
|
|
124
|
+
The sidebar lists every table with its row count. Click a column header to sort, type in the filter box to narrow rows (`/` jumps to it), click a row to see a long formula or SQL statement in full. The tiles on the overview open their tables. The graph view lets you copy or download the DOT text. The Field renames view has the buttons for the feature above: pick a style, datasource and optional reference workbook, then **Create fixed workbook** saves `<name>_renamed` beside the original (and says so if that file already exists) or **Download fixed workbook** sends it to your browser without saving anything. Everything else exports as CSV. It has a dark theme and works in a narrow window.
|
|
125
|
+
|
|
126
|
+

|
|
127
|
+
|
|
128
|
+
It's meant to run on your own machine for one person. It refuses requests that come from other websites, but there's no login, so don't put it on a shared network.
|
|
129
|
+
|
|
130
|
+
## Tests
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
pip install -e ".[test]"
|
|
134
|
+
pytest
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
|
|
138
|
+
|
|
139
|
+
The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
pip install -e ".[test,browser]"
|
|
143
|
+
playwright install chromium # add --with-deps if you have root
|
|
144
|
+
./scripts/setup-browser-libs.sh # without root, this unpacks the system libraries locally
|
|
145
|
+
pytest tests/test_gui_browser.py
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
## Compared with the R package
|
|
149
|
+
|
|
150
|
+
The code follows the R original function by function, and the aim is the same output. Not everything is ported. Missing so far: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, the analytics helpers (calculation complexity, field usage, replication brief) and the Shiny inspector. The GUI covers some of what the inspector did.
|
|
151
|
+
|
|
152
|
+
Where the R version has a bug, this one doesn't copy it. That currently covers joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
|
|
153
|
+
|
|
154
|
+
## Why a Tableau parser
|
|
155
|
+
|
|
156
|
+
A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python isn't hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
|
|
157
|
+
|
|
158
|
+
## Credit
|
|
159
|
+
|
|
160
|
+
Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
|
|
Binary file
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""
|
|
1
|
+
"""py_tbparse: a native Python port of the twbparser R package.
|
|
2
2
|
|
|
3
3
|
Parses Tableau .twb/.twbx workbook files into pandas DataFrames. Ported
|
|
4
4
|
from https://github.com/PrigasG/twbparser (MIT licensed).
|
|
@@ -15,6 +15,13 @@ from .joins import extract_joins
|
|
|
15
15
|
from .parser import TwbParser
|
|
16
16
|
from .published import extract_published_refs
|
|
17
17
|
from .relationships import extract_relations, extract_relationships
|
|
18
|
+
from .rename import (
|
|
19
|
+
apply_field_renames,
|
|
20
|
+
compare_field_schemas,
|
|
21
|
+
load_rename_mapping,
|
|
22
|
+
normalize_name,
|
|
23
|
+
suggest_field_renames,
|
|
24
|
+
)
|
|
18
25
|
from .sql import extract_custom_sql, extract_initial_sql
|
|
19
26
|
from .validators import validate_relationships
|
|
20
27
|
from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
|
|
@@ -44,6 +51,11 @@ __all__ = [
|
|
|
44
51
|
"diff_tables",
|
|
45
52
|
"diff_workbooks",
|
|
46
53
|
"scan_folder",
|
|
54
|
+
"apply_field_renames",
|
|
55
|
+
"compare_field_schemas",
|
|
56
|
+
"load_rename_mapping",
|
|
57
|
+
"normalize_name",
|
|
58
|
+
"suggest_field_renames",
|
|
47
59
|
]
|
|
48
60
|
|
|
49
61
|
try:
|
|
@@ -38,6 +38,13 @@ def _calculated_fields(p: TwbParser, include_parameters: bool = False, **_kw) ->
|
|
|
38
38
|
return p.get_calculated_fields(include_parameters=include_parameters)
|
|
39
39
|
|
|
40
40
|
|
|
41
|
+
def _field_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
|
|
42
|
+
datasource=None, **_kw) -> pd.DataFrame:
|
|
43
|
+
return p.get_field_renames(
|
|
44
|
+
reference=reference, style=style, only_changed=only_changed, datasource=datasource
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
|
|
41
48
|
def _joins(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
42
49
|
return p.get_joins()
|
|
43
50
|
|
|
@@ -83,6 +90,7 @@ TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
|
|
|
83
90
|
"fields": _fields,
|
|
84
91
|
"raw-fields": _raw_fields,
|
|
85
92
|
"calculated-fields": _calculated_fields,
|
|
93
|
+
"field-renames": _field_renames,
|
|
86
94
|
"joins": _joins,
|
|
87
95
|
"relations": _relations,
|
|
88
96
|
"relationships": _relationships,
|