py-tbparse 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/AGENTS.md +10 -9
  2. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/LICENSE +1 -1
  3. py_tbparse-0.3.0/PKG-INFO +196 -0
  4. py_tbparse-0.3.0/README.md +160 -0
  5. py_tbparse-0.3.0/docs/gui-field-renames.png +0 -0
  6. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/__init__.py +13 -1
  7. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/_tables.py +8 -0
  8. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/cli.py +114 -8
  9. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/parser.py +14 -0
  10. py_tbparse-0.3.0/py_tbparse/rename.py +520 -0
  11. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/webgui.py +198 -15
  12. py_tbparse-0.3.0/py_tbparse.egg-info/PKG-INFO +196 -0
  13. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/py_tbparse.egg-info/SOURCES.txt +26 -21
  14. py_tbparse-0.3.0/py_tbparse.egg-info/entry_points.txt +3 -0
  15. py_tbparse-0.3.0/py_tbparse.egg-info/top_level.txt +1 -0
  16. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/pyproject.toml +5 -5
  17. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_batch.py +1 -1
  18. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_calculated_fields.py +1 -1
  19. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_clean.py +1 -1
  20. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_cli.py +1 -1
  21. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_dashboards.py +2 -2
  22. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_datasources.py +2 -2
  23. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_diff.py +1 -1
  24. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_fields.py +1 -1
  25. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_graph.py +1 -1
  26. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_gui_browser.py +49 -1
  27. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_joins.py +1 -1
  28. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_parser.py +2 -2
  29. py_tbparse-0.3.0/tests/test_public_workbooks.py +31 -0
  30. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_published.py +1 -1
  31. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_relationships.py +3 -3
  32. py_tbparse-0.3.0/tests/test_rename.py +386 -0
  33. py_tbparse-0.3.0/tests/test_rename_mapping.py +68 -0
  34. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_sql.py +1 -1
  35. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_validators.py +2 -2
  36. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_version.py +3 -3
  37. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_webgui.py +80 -2
  38. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/test_xml.py +2 -2
  39. py_tbparse-0.2.0/PKG-INFO +0 -150
  40. py_tbparse-0.2.0/README.md +0 -114
  41. py_tbparse-0.2.0/py_tbparse.egg-info/PKG-INFO +0 -150
  42. py_tbparse-0.2.0/py_tbparse.egg-info/entry_points.txt +0 -3
  43. py_tbparse-0.2.0/py_tbparse.egg-info/top_level.txt +0 -1
  44. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/MANIFEST.in +0 -0
  45. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/docs/gui-overview.png +0 -0
  46. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/_clean.py +0 -0
  47. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/_xml.py +0 -0
  48. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/batch.py +0 -0
  49. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/calculated_fields.py +0 -0
  50. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/dashboards.py +0 -0
  51. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/datasources.py +0 -0
  52. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/diff.py +0 -0
  53. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/fields.py +0 -0
  54. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/graph.py +0 -0
  55. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/joins.py +0 -0
  56. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/published.py +0 -0
  57. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/py.typed +0 -0
  58. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/relationships.py +0 -0
  59. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/sql.py +0 -0
  60. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.3.0/py_tbparse}/validators.py +0 -0
  61. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/py_tbparse.egg-info/dependency_links.txt +0 -0
  62. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/py_tbparse.egg-info/requires.txt +0 -0
  63. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/setup.cfg +0 -0
  64. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/conftest.py +0 -0
  65. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/fixtures/test_for_wenjie.twb +0 -0
  66. {py_tbparse-0.2.0 → py_tbparse-0.3.0}/tests/fixtures/test_for_zip.twbx +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## What this is
4
4
 
5
- `twbparser_py` is a native Python port of the R package
5
+ `py_tbparse` is a native Python port of the R package
6
6
  [`twbparser`](https://github.com/PrigasG/twbparser) (mirrored at
7
7
  `DDSNA/twbparser`): it parses Tableau `.twb`/`.twbx` workbook files into
8
8
  `pandas` DataFrames. Pure `lxml` XML parsing — no R runtime, no `rpy2`.
@@ -37,7 +37,7 @@ python3 -m venv .venv
37
37
  ```bash
38
38
  .venv/bin/python -m pytest -q # run the full test suite
39
39
  .venv/bin/python -m pytest -q tests/test_joins.py # single file
40
- .venv/bin/python -m py_compile twbparser_py/*.py # syntax check
40
+ .venv/bin/python -m py_compile py_tbparse/*.py # syntax check
41
41
  ```
42
42
 
43
43
  Browser (GUI) tests need a one-time setup; without it they skip and the
@@ -58,19 +58,19 @@ via `from __future__ import annotations`).
58
58
 
59
59
  ```bash
60
60
  .venv/bin/pip install -e ".[dev]" # build + twine
61
- rm -rf dist build twbparser_py.egg-info
61
+ rm -rf dist build py_tbparse.egg-info
62
62
  .venv/bin/python -m build # produces dist/*.whl and dist/*.tar.gz
63
63
  .venv/bin/twine check dist/* # validates metadata/README rendering
64
64
  ```
65
65
 
66
66
  Version is single-sourced from `pyproject.toml`'s `[project].version`;
67
- `twbparser_py.__version__` reads it back via `importlib.metadata` at
67
+ `py_tbparse.__version__` reads it back via `importlib.metadata` at
68
68
  runtime (see `__init__.py`), so don't hardcode a second copy.
69
69
 
70
70
  Before bumping the version for a release: bump `version` in
71
71
  `pyproject.toml`, then rebuild and smoke-test the wheel in a
72
- throwaway venv (`pip install dist/*.whl`, run `twbparser --help` and
73
- `twbparser-gui --help`, run pytest against an extracted sdist) — this
72
+ throwaway venv (`pip install dist/*.whl`, run `py-tbparse --help` and
73
+ `py-tbparse-gui --help`, run pytest against an extracted sdist) — this
74
74
  catches packaging bugs (missing files, wrong entry points) that an
75
75
  editable install won't.
76
76
 
@@ -88,7 +88,7 @@ Publishers" settings page before the first release, and needs a
88
88
 
89
89
  ## Architecture
90
90
 
91
- Each `twbparser_py/*.py` module is a direct port of one R source file in
91
+ Each `py_tbparse/*.py` module is a direct port of one R source file in
92
92
  the upstream package, function-for-function:
93
93
 
94
94
  | Python module | Ported from (R) | Notes |
@@ -120,10 +120,11 @@ beyond the R package's scope:
120
120
  | Python module | What it is |
121
121
  |---|---|
122
122
  | `_tables.py` | Name → `TwbParser`-accessor registry shared by `cli.py`, `webgui.py`, `diff.py`, and `batch.py`. Adding a new extractor to `TwbParser`? Add it here too so it's automatically available everywhere else. |
123
- | `cli.py` | `twbparser` command-line entry point, plus the `diff`/`batch` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
124
- | `webgui.py` | `twbparser-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
123
+ | `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
124
+ | `webgui.py` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
125
125
  | `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
126
126
  | `diff.py` | `diff_tables()`/`diff_workbooks()`: row-level added/removed diff between two workbooks' same-named table, via `_tables.TABLE_SPECS`. No "changed" classification without a natural key — a changed row shows as one removed + one added row. |
127
+ | `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` table in `_tables.py`. |
127
128
  | `batch.py` | `scan_folder()`: runs one table across every `.twb`/`.twbx` in a directory, concatenated with a `workbook` column. Skips (with a warning) any file that fails to load/extract rather than aborting the batch. |
128
129
 
129
130
  ## Porting conventions (read before adding/modifying a function)
@@ -4,7 +4,7 @@ This project is a Python port of logic originally implemented in the R
4
4
  package "twbparser" (https://github.com/PrigasG/twbparser),
5
5
  Copyright (c) 2025 George Arthur.
6
6
 
7
- Copyright (c) 2026 the twbparser-py contributors
7
+ Copyright (c) 2026 the py-tbparse contributors
8
8
 
9
9
  Permission is hereby granted, free of charge, to any person obtaining a copy
10
10
  of this software and associated documentation files (the "Software"), to deal
@@ -0,0 +1,196 @@
1
+ Metadata-Version: 2.4
2
+ Name: py-tbparse
3
+ Version: 0.3.0
4
+ Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
5
+ Author: DDSNA
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/DDSNA/py-tbparse
8
+ Project-URL: Source, https://github.com/DDSNA/py-tbparse
9
+ Project-URL: Issues, https://github.com/DDSNA/py-tbparse/issues
10
+ Keywords: tableau,twb,twbx,workbook,parser,pandas
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
+ Classifier: Topic :: Office/Business
23
+ Requires-Python: >=3.9
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: lxml>=4.9
27
+ Requires-Dist: pandas>=1.5
28
+ Provides-Extra: test
29
+ Requires-Dist: pytest>=7; extra == "test"
30
+ Provides-Extra: browser
31
+ Requires-Dist: playwright>=1.40; extra == "browser"
32
+ Provides-Extra: dev
33
+ Requires-Dist: build>=1.0; extra == "dev"
34
+ Requires-Dist: twine>=5.0; extra == "dev"
35
+ Dynamic: license-file
36
+
37
+ # py-tbparse
38
+
39
+ Reads Tableau workbooks (`.twb` and `.twbx`) and gives you what's in them as pandas DataFrames: datasources, fields, calculated fields, joins, relationships, dashboards, custom SQL. It's plain Python. You don't need Tableau or R installed.
40
+
41
+ It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The browser GUI, the command-line tool, workbook diffing and folder scanning are new here.
42
+
43
+ ![py-tbparse GUI, overview of a loaded workbook](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-overview.png)
44
+
45
+ ## Install
46
+
47
+ ```bash
48
+ pip install py-tbparse
49
+ ```
50
+
51
+ Or from a checkout:
52
+
53
+ ```bash
54
+ git clone https://github.com/DDSNA/py-tbparse.git
55
+ cd py-tbparse
56
+ pip install -e .
57
+ ```
58
+
59
+ One name throughout: `pip install py-tbparse`, `import py_tbparse`, run `py-tbparse` or `py-tbparse-gui`. (Releases up to 0.2.0 imported `twbparser_py` and ran `twbparser` / `twbparser-gui`; those names are gone.)
60
+
61
+ ## Using it from Python
62
+
63
+ ```python
64
+ from py_tbparse import TwbParser
65
+
66
+ p = TwbParser("workbook.twb") # or a .twbx
67
+ p.get_overview() # counts of everything
68
+ p.get_calculated_fields() # each calculation and its formula
69
+ p.get_relationships() # how the tables connect
70
+ ```
71
+
72
+ Every getter returns a DataFrame. The rest are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql` and `get_published_refs`. `get_relationship_graph_dot()` returns the data model as Graphviz text, and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
73
+
74
+ Two helpers work across workbooks:
75
+
76
+ ```python
77
+ from py_tbparse import diff_workbooks, scan_folder
78
+
79
+ diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
80
+ scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
81
+ ```
82
+
83
+ ### Cleaning field names after a datasource switch
84
+
85
+ Pointing a workbook at a new datasource that only partly matches the old schema tends to leave ugly names: `ORDER_ID`, `orderId`, `Order ID (Orders1)`, `Order ID1`. `suggest_field_renames` proposes a clean name for each field. It only reports; it never edits the workbook.
86
+
87
+ ```python
88
+ from py_tbparse import TwbParser, suggest_field_renames
89
+
90
+ new = TwbParser("after_switch.twb")
91
+ old = TwbParser("before_switch.twb") # optional: the schema you want to match
92
+
93
+ suggest_field_renames(new, reference=old, only_changed=True)
94
+ # name current suggested reason score
95
+ # [ORDER_ID] ORDER_ID Order ID matches reference 1.0
96
+ # [Sales Amount (Orders1)] Sales Amount (Orders1) Sales Amount matches reference 1.0
97
+ # [orderDate] orderDate Order Date normalized NaN
98
+ ```
99
+
100
+ With a `reference` (a workbook, a fields table or a plain list of names), fields that match by name ignoring case, separators and Tableau's duplicate suffixes take the reference's exact spelling, and near-misses above `fuzzy_cutoff` (default 0.85) are matched too. Everything else is tidied by `normalize_name(name, style)`, where `style` is `title` (default), `snake`, `lower` or `keep`. Names that are already clean (`YTD Sales`, `iPhone Units`, `Country/Region`) are left alone. `1` and `(Table1)` suffixes are only dropped when the plain name exists in the same datasource (`Address Line 2` and `Q1` are never treated as duplicates), fuzzy matches never cross a different number, and two fields never get the same suggestion (the loser stays as it is, with `reason` set to `conflict`).
101
+
102
+ `p.get_field_renames()` does the same from a parser.
103
+
104
+ **Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
105
+
106
+ ```bash
107
+ py-tbparse rename new.twb -r old.twb -f csv -o mapping.csv
108
+ py-tbparse rename new.twb --apply mapping.csv # writes new_renamed.twb; add --write-workbook PATH to choose the file
109
+ ```
110
+
111
+ Your edits are applied as written, including rows the tool had marked `conflict`; two rows giving the same name in one datasource are rejected. From Python this is `apply_field_renames(p, renames=load_rename_mapping("mapping.csv"))`.
112
+
113
+ **What will stay broken.** `py-tbparse rename new.twb -r old.twb --missing` (or `compare_field_schemas(new, old)`) lists the fields with no counterpart after the switch: `old only` fields that nothing in the new source matches, and `new only` fields nothing in the old workbook matches, each with the closest name on the other side as a hint. Sheets using an `old only` field stay red after Replace Data Source until you map or recreate it.
114
+
115
+ To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...)`) writes `<name>_renamed.twb` / `.twbx` next to the original. It sets each field's caption, which is how Tableau renames a field; the internal names that formulas and sheets use are not touched, and a `.twbx` keeps all its other contents. A field that only exists as a physical column (typical right after a datasource switch) gets a new minimal `<column>` element carrying the caption; that shape follows what Tableau writes but I have not opened such files in Tableau itself. It never modifies the original and refuses to overwrite an existing file unless you pass `overwrite=True`.
116
+
117
+ **When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
118
+
119
+ ### `.twbx` files
120
+
121
+ A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
122
+
123
+ ```python
124
+ from py_tbparse import extract_twb_from_twbx, twbx_extract_files
125
+
126
+ extract_twb_from_twbx("workbook.twbx", extract_dir="out/")
127
+ twbx_extract_files("workbook.twbx", exdir="out/") # everything in the package
128
+ ```
129
+
130
+ ## Command line
131
+
132
+ ```bash
133
+ py-tbparse workbook.twb # overview
134
+ py-tbparse workbook.twb tables # list the tables
135
+ py-tbparse workbook.twb calculated-fields
136
+ py-tbparse workbook.twb fields --format csv -o fields.csv
137
+ py-tbparse workbook.twbx dashboard-sheets --dashboard "Sales Overview"
138
+ py-tbparse workbook.twb validate # exit code 2 if it finds a problem
139
+ py-tbparse workbook.twb graph > model.dot # --include-inferred adds the guessed links
140
+ py-tbparse diff old.twb new.twb datasources
141
+ py-tbparse batch ./workbooks datasources
142
+ py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
143
+ py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
144
+ ```
145
+
146
+ Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`.
147
+
148
+ `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
149
+
150
+ ## GUI
151
+
152
+ ```bash
153
+ py-tbparse-gui workbook.twb # opens your browser with it loaded
154
+ py-tbparse-gui # starts empty, paste a path and hit Load
155
+ py-tbparse-gui --no-browser --port 8765 # server only, for a machine with no display
156
+ ```
157
+
158
+ It uses only the standard library, so there's nothing more to install.
159
+
160
+ The sidebar lists every table with its row count. Click a column header to sort, type in the filter box to narrow rows (`/` jumps to it), click a row to see a long formula or SQL statement in full. The tiles on the overview open their tables. The graph view lets you copy or download the DOT text. The Field renames view has the buttons for the feature above: pick a style, datasource and optional reference workbook, then **Create fixed workbook** saves `<name>_renamed` beside the original (and says so if that file already exists) or **Download fixed workbook** sends it to your browser without saving anything. Everything else exports as CSV. It has a dark theme and works in a narrow window.
161
+
162
+ ![py-tbparse GUI, Field renames view with suggested clean names](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-field-renames.png)
163
+
164
+ It's meant to run on your own machine for one person. It refuses requests that come from other websites, but there's no login, so don't put it on a shared network.
165
+
166
+ ## Tests
167
+
168
+ ```bash
169
+ pip install -e ".[test]"
170
+ pytest
171
+ ```
172
+
173
+ The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
174
+
175
+ The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
176
+
177
+ ```bash
178
+ pip install -e ".[test,browser]"
179
+ playwright install chromium # add --with-deps if you have root
180
+ ./scripts/setup-browser-libs.sh # without root, this unpacks the system libraries locally
181
+ pytest tests/test_gui_browser.py
182
+ ```
183
+
184
+ ## Compared with the R package
185
+
186
+ The code follows the R original function by function, and the aim is the same output. Not everything is ported. Missing so far: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, the analytics helpers (calculation complexity, field usage, replication brief) and the Shiny inspector. The GUI covers some of what the inspector did.
187
+
188
+ Where the R version has a bug, this one doesn't copy it. That currently covers joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
189
+
190
+ ## Why a Tableau parser
191
+
192
+ A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python isn't hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
193
+
194
+ ## Credit
195
+
196
+ Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
@@ -0,0 +1,160 @@
1
+ # py-tbparse
2
+
3
+ Reads Tableau workbooks (`.twb` and `.twbx`) and gives you what's in them as pandas DataFrames: datasources, fields, calculated fields, joins, relationships, dashboards, custom SQL. It's plain Python. You don't need Tableau or R installed.
4
+
5
+ It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The browser GUI, the command-line tool, workbook diffing and folder scanning are new here.
6
+
7
+ ![py-tbparse GUI, overview of a loaded workbook](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-overview.png)
8
+
9
+ ## Install
10
+
11
+ ```bash
12
+ pip install py-tbparse
13
+ ```
14
+
15
+ Or from a checkout:
16
+
17
+ ```bash
18
+ git clone https://github.com/DDSNA/py-tbparse.git
19
+ cd py-tbparse
20
+ pip install -e .
21
+ ```
22
+
23
+ One name throughout: `pip install py-tbparse`, `import py_tbparse`, run `py-tbparse` or `py-tbparse-gui`. (Releases up to 0.2.0 imported `twbparser_py` and ran `twbparser` / `twbparser-gui`; those names are gone.)
24
+
25
+ ## Using it from Python
26
+
27
+ ```python
28
+ from py_tbparse import TwbParser
29
+
30
+ p = TwbParser("workbook.twb") # or a .twbx
31
+ p.get_overview() # counts of everything
32
+ p.get_calculated_fields() # each calculation and its formula
33
+ p.get_relationships() # how the tables connect
34
+ ```
35
+
36
+ Every getter returns a DataFrame. The rest are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql` and `get_published_refs`. `get_relationship_graph_dot()` returns the data model as Graphviz text, and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
37
+
38
+ Two helpers work across workbooks:
39
+
40
+ ```python
41
+ from py_tbparse import diff_workbooks, scan_folder
42
+
43
+ diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
44
+ scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
45
+ ```
46
+
47
+ ### Cleaning field names after a datasource switch
48
+
49
+ Pointing a workbook at a new datasource that only partly matches the old schema tends to leave ugly names: `ORDER_ID`, `orderId`, `Order ID (Orders1)`, `Order ID1`. `suggest_field_renames` proposes a clean name for each field. It only reports; it never edits the workbook.
50
+
51
+ ```python
52
+ from py_tbparse import TwbParser, suggest_field_renames
53
+
54
+ new = TwbParser("after_switch.twb")
55
+ old = TwbParser("before_switch.twb") # optional: the schema you want to match
56
+
57
+ suggest_field_renames(new, reference=old, only_changed=True)
58
+ # name current suggested reason score
59
+ # [ORDER_ID] ORDER_ID Order ID matches reference 1.0
60
+ # [Sales Amount (Orders1)] Sales Amount (Orders1) Sales Amount matches reference 1.0
61
+ # [orderDate] orderDate Order Date normalized NaN
62
+ ```
63
+
64
+ With a `reference` (a workbook, a fields table or a plain list of names), fields that match by name ignoring case, separators and Tableau's duplicate suffixes take the reference's exact spelling, and near-misses above `fuzzy_cutoff` (default 0.85) are matched too. Everything else is tidied by `normalize_name(name, style)`, where `style` is `title` (default), `snake`, `lower` or `keep`. Names that are already clean (`YTD Sales`, `iPhone Units`, `Country/Region`) are left alone. `1` and `(Table1)` suffixes are only dropped when the plain name exists in the same datasource (`Address Line 2` and `Q1` are never treated as duplicates), fuzzy matches never cross a different number, and two fields never get the same suggestion (the loser stays as it is, with `reason` set to `conflict`).
65
+
66
+ `p.get_field_renames()` does the same from a parser.
67
+
68
+ **Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
69
+
70
+ ```bash
71
+ py-tbparse rename new.twb -r old.twb -f csv -o mapping.csv
72
+ py-tbparse rename new.twb --apply mapping.csv # writes new_renamed.twb; add --write-workbook PATH to choose the file
73
+ ```
74
+
75
+ Your edits are applied as written, including rows the tool had marked `conflict`; two rows giving the same name in one datasource are rejected. From Python this is `apply_field_renames(p, renames=load_rename_mapping("mapping.csv"))`.
76
+
77
+ **What will stay broken.** `py-tbparse rename new.twb -r old.twb --missing` (or `compare_field_schemas(new, old)`) lists the fields with no counterpart after the switch: `old only` fields that nothing in the new source matches, and `new only` fields nothing in the old workbook matches, each with the closest name on the other side as a hint. Sheets using an `old only` field stay red after Replace Data Source until you map or recreate it.
78
+
79
+ To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...)`) writes `<name>_renamed.twb` / `.twbx` next to the original. It sets each field's caption, which is how Tableau renames a field; the internal names that formulas and sheets use are not touched, and a `.twbx` keeps all its other contents. A field that only exists as a physical column (typical right after a datasource switch) gets a new minimal `<column>` element carrying the caption; that shape follows what Tableau writes but I have not opened such files in Tableau itself. It never modifies the original and refuses to overwrite an existing file unless you pass `overwrite=True`.
80
+
81
+ **When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
82
+
83
+ ### `.twbx` files
84
+
85
+ A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
86
+
87
+ ```python
88
+ from py_tbparse import extract_twb_from_twbx, twbx_extract_files
89
+
90
+ extract_twb_from_twbx("workbook.twbx", extract_dir="out/")
91
+ twbx_extract_files("workbook.twbx", exdir="out/") # everything in the package
92
+ ```
93
+
94
+ ## Command line
95
+
96
+ ```bash
97
+ py-tbparse workbook.twb # overview
98
+ py-tbparse workbook.twb tables # list the tables
99
+ py-tbparse workbook.twb calculated-fields
100
+ py-tbparse workbook.twb fields --format csv -o fields.csv
101
+ py-tbparse workbook.twbx dashboard-sheets --dashboard "Sales Overview"
102
+ py-tbparse workbook.twb validate # exit code 2 if it finds a problem
103
+ py-tbparse workbook.twb graph > model.dot # --include-inferred adds the guessed links
104
+ py-tbparse diff old.twb new.twb datasources
105
+ py-tbparse batch ./workbooks datasources
106
+ py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
107
+ py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
108
+ ```
109
+
110
+ Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`.
111
+
112
+ `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
113
+
114
+ ## GUI
115
+
116
+ ```bash
117
+ py-tbparse-gui workbook.twb # opens your browser with it loaded
118
+ py-tbparse-gui # starts empty, paste a path and hit Load
119
+ py-tbparse-gui --no-browser --port 8765 # server only, for a machine with no display
120
+ ```
121
+
122
+ It uses only the standard library, so there's nothing more to install.
123
+
124
+ The sidebar lists every table with its row count. Click a column header to sort, type in the filter box to narrow rows (`/` jumps to it), click a row to see a long formula or SQL statement in full. The tiles on the overview open their tables. The graph view lets you copy or download the DOT text. The Field renames view has the buttons for the feature above: pick a style, datasource and optional reference workbook, then **Create fixed workbook** saves `<name>_renamed` beside the original (and says so if that file already exists) or **Download fixed workbook** sends it to your browser without saving anything. Everything else exports as CSV. It has a dark theme and works in a narrow window.
125
+
126
+ ![py-tbparse GUI, Field renames view with suggested clean names](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-field-renames.png)
127
+
128
+ It's meant to run on your own machine for one person. It refuses requests that come from other websites, but there's no login, so don't put it on a shared network.
129
+
130
+ ## Tests
131
+
132
+ ```bash
133
+ pip install -e ".[test]"
134
+ pytest
135
+ ```
136
+
137
+ The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
138
+
139
+ The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
140
+
141
+ ```bash
142
+ pip install -e ".[test,browser]"
143
+ playwright install chromium # add --with-deps if you have root
144
+ ./scripts/setup-browser-libs.sh # without root, this unpacks the system libraries locally
145
+ pytest tests/test_gui_browser.py
146
+ ```
147
+
148
+ ## Compared with the R package
149
+
150
+ The code follows the R original function by function, and the aim is the same output. Not everything is ported. Missing so far: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, the analytics helpers (calculation complexity, field usage, replication brief) and the Shiny inspector. The GUI covers some of what the inspector did.
151
+
152
+ Where the R version has a bug, this one doesn't copy it. That currently covers joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
153
+
154
+ ## Why a Tableau parser
155
+
156
+ A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python isn't hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
157
+
158
+ ## Credit
159
+
160
+ Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
@@ -1,4 +1,4 @@
1
- """twbparser_py: a native Python port of the twbparser R package.
1
+ """py_tbparse: a native Python port of the twbparser R package.
2
2
 
3
3
  Parses Tableau .twb/.twbx workbook files into pandas DataFrames. Ported
4
4
  from https://github.com/PrigasG/twbparser (MIT licensed).
@@ -15,6 +15,13 @@ from .joins import extract_joins
15
15
  from .parser import TwbParser
16
16
  from .published import extract_published_refs
17
17
  from .relationships import extract_relations, extract_relationships
18
+ from .rename import (
19
+ apply_field_renames,
20
+ compare_field_schemas,
21
+ load_rename_mapping,
22
+ normalize_name,
23
+ suggest_field_renames,
24
+ )
18
25
  from .sql import extract_custom_sql, extract_initial_sql
19
26
  from .validators import validate_relationships
20
27
  from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
@@ -44,6 +51,11 @@ __all__ = [
44
51
  "diff_tables",
45
52
  "diff_workbooks",
46
53
  "scan_folder",
54
+ "apply_field_renames",
55
+ "compare_field_schemas",
56
+ "load_rename_mapping",
57
+ "normalize_name",
58
+ "suggest_field_renames",
47
59
  ]
48
60
 
49
61
  try:
@@ -38,6 +38,13 @@ def _calculated_fields(p: TwbParser, include_parameters: bool = False, **_kw) ->
38
38
  return p.get_calculated_fields(include_parameters=include_parameters)
39
39
 
40
40
 
41
+ def _field_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
42
+ datasource=None, **_kw) -> pd.DataFrame:
43
+ return p.get_field_renames(
44
+ reference=reference, style=style, only_changed=only_changed, datasource=datasource
45
+ )
46
+
47
+
41
48
  def _joins(p: TwbParser, **_kw) -> pd.DataFrame:
42
49
  return p.get_joins()
43
50
 
@@ -83,6 +90,7 @@ TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
83
90
  "fields": _fields,
84
91
  "raw-fields": _raw_fields,
85
92
  "calculated-fields": _calculated_fields,
93
+ "field-renames": _field_renames,
86
94
  "joins": _joins,
87
95
  "relations": _relations,
88
96
  "relationships": _relationships,