py-tbparse 0.2.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/AGENTS.md +21 -11
  2. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/LICENSE +1 -1
  3. py_tbparse-0.4.0/PKG-INFO +250 -0
  4. py_tbparse-0.4.0/README.md +214 -0
  5. py_tbparse-0.4.0/docs/gui-field-renames.png +0 -0
  6. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/__init__.py +35 -1
  7. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/_tables.py +26 -0
  8. py_tbparse-0.4.0/py_tbparse/cli.py +436 -0
  9. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/parser.py +29 -0
  10. py_tbparse-0.4.0/py_tbparse/rename.py +759 -0
  11. py_tbparse-0.4.0/py_tbparse/templates.py +855 -0
  12. py_tbparse-0.4.0/py_tbparse/usage.py +166 -0
  13. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/webgui.py +211 -15
  14. py_tbparse-0.4.0/py_tbparse.egg-info/PKG-INFO +250 -0
  15. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/SOURCES.txt +31 -21
  16. py_tbparse-0.4.0/py_tbparse.egg-info/entry_points.txt +3 -0
  17. py_tbparse-0.4.0/py_tbparse.egg-info/top_level.txt +1 -0
  18. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/pyproject.toml +5 -5
  19. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_batch.py +1 -1
  20. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_calculated_fields.py +1 -1
  21. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_clean.py +1 -1
  22. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_cli.py +1 -1
  23. py_tbparse-0.4.0/tests/test_corpus.py +84 -0
  24. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_dashboards.py +2 -2
  25. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_datasources.py +2 -2
  26. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_diff.py +1 -1
  27. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_fields.py +1 -1
  28. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_graph.py +1 -1
  29. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_gui_browser.py +72 -1
  30. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_joins.py +1 -1
  31. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_parser.py +2 -2
  32. py_tbparse-0.4.0/tests/test_public_workbooks.py +49 -0
  33. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_published.py +1 -1
  34. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_relationships.py +3 -3
  35. py_tbparse-0.4.0/tests/test_rename.py +386 -0
  36. py_tbparse-0.4.0/tests/test_rename_all.py +215 -0
  37. py_tbparse-0.4.0/tests/test_rename_mapping.py +68 -0
  38. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_sql.py +1 -1
  39. py_tbparse-0.4.0/tests/test_templates.py +415 -0
  40. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_validators.py +2 -2
  41. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_version.py +3 -3
  42. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_webgui.py +105 -2
  43. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/test_xml.py +2 -2
  44. py_tbparse-0.2.0/PKG-INFO +0 -150
  45. py_tbparse-0.2.0/README.md +0 -114
  46. py_tbparse-0.2.0/py_tbparse.egg-info/PKG-INFO +0 -150
  47. py_tbparse-0.2.0/py_tbparse.egg-info/entry_points.txt +0 -3
  48. py_tbparse-0.2.0/py_tbparse.egg-info/top_level.txt +0 -1
  49. py_tbparse-0.2.0/twbparser_py/cli.py +0 -207
  50. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/MANIFEST.in +0 -0
  51. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/docs/gui-overview.png +0 -0
  52. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/_clean.py +0 -0
  53. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/_xml.py +0 -0
  54. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/batch.py +0 -0
  55. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/calculated_fields.py +0 -0
  56. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/dashboards.py +0 -0
  57. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/datasources.py +0 -0
  58. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/diff.py +0 -0
  59. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/fields.py +0 -0
  60. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/graph.py +0 -0
  61. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/joins.py +0 -0
  62. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/published.py +0 -0
  63. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/py.typed +0 -0
  64. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/relationships.py +0 -0
  65. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/sql.py +0 -0
  66. {py_tbparse-0.2.0/twbparser_py → py_tbparse-0.4.0/py_tbparse}/validators.py +0 -0
  67. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/dependency_links.txt +0 -0
  68. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/requires.txt +0 -0
  69. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/setup.cfg +0 -0
  70. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/conftest.py +0 -0
  71. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/fixtures/test_for_wenjie.twb +0 -0
  72. {py_tbparse-0.2.0 → py_tbparse-0.4.0}/tests/fixtures/test_for_zip.twbx +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## What this is
4
4
 
5
- `twbparser_py` is a native Python port of the R package
5
+ `py_tbparse` is a native Python port of the R package
6
6
  [`twbparser`](https://github.com/PrigasG/twbparser) (mirrored at
7
7
  `DDSNA/twbparser`): it parses Tableau `.twb`/`.twbx` workbook files into
8
8
  `pandas` DataFrames. Pure `lxml` XML parsing — no R runtime, no `rpy2`.
@@ -14,14 +14,15 @@ type="join">` and 2020.2+ `<relationships>`), inferred relationships,
14
14
  dashboards/dashboard-sheets, `validate_relationships`, custom/initial SQL,
15
15
  and published-source detection. **Not yet ported** (v2): formatting,
16
16
  tooltips, colors, axes, sorts, dashboard layout/actions, analytics
17
- helpers (calc complexity, field usage, replication brief), and the
17
+ helpers (calc complexity, replication brief), and the
18
18
  Shiny-inspector equivalent.
19
19
 
20
20
  Also added, with no R equivalent (Python-native extras -- see "Non-R
21
21
  modules" below): Graphviz DOT export of the relationship graph
22
22
  (replaces the R package's igraph/ggraph-based plotting with a
23
23
  dependency-free alternative), workbook-to-workbook diff, folder/batch
24
- analysis across many workbooks, and Jupyter rich display.
24
+ analysis across many workbooks, field renaming, field usage, workbook
25
+ templates, and Jupyter rich display.
25
26
 
26
27
  ## Setup
27
28
 
@@ -37,7 +38,7 @@ python3 -m venv .venv
37
38
  ```bash
38
39
  .venv/bin/python -m pytest -q # run the full test suite
39
40
  .venv/bin/python -m pytest -q tests/test_joins.py # single file
40
- .venv/bin/python -m py_compile twbparser_py/*.py # syntax check
41
+ .venv/bin/python -m py_compile py_tbparse/*.py # syntax check
41
42
  ```
42
43
 
43
44
  Browser (GUI) tests need a one-time setup; without it they skip and the
@@ -58,19 +59,19 @@ via `from __future__ import annotations`).
58
59
 
59
60
  ```bash
60
61
  .venv/bin/pip install -e ".[dev]" # build + twine
61
- rm -rf dist build twbparser_py.egg-info
62
+ rm -rf dist build py_tbparse.egg-info
62
63
  .venv/bin/python -m build # produces dist/*.whl and dist/*.tar.gz
63
64
  .venv/bin/twine check dist/* # validates metadata/README rendering
64
65
  ```
65
66
 
66
67
  Version is single-sourced from `pyproject.toml`'s `[project].version`;
67
- `twbparser_py.__version__` reads it back via `importlib.metadata` at
68
+ `py_tbparse.__version__` reads it back via `importlib.metadata` at
68
69
  runtime (see `__init__.py`), so don't hardcode a second copy.
69
70
 
70
71
  Before bumping the version for a release: bump `version` in
71
72
  `pyproject.toml`, then rebuild and smoke-test the wheel in a
72
- throwaway venv (`pip install dist/*.whl`, run `twbparser --help` and
73
- `twbparser-gui --help`, run pytest against an extracted sdist) — this
73
+ throwaway venv (`pip install dist/*.whl`, run `py-tbparse --help` and
74
+ `py-tbparse-gui --help`, run pytest against an extracted sdist) — this
74
75
  catches packaging bugs (missing files, wrong entry points) that an
75
76
  editable install won't.
76
77
 
@@ -88,7 +89,7 @@ Publishers" settings page before the first release, and needs a
88
89
 
89
90
  ## Architecture
90
91
 
91
- Each `twbparser_py/*.py` module is a direct port of one R source file in
92
+ Each `py_tbparse/*.py` module is a direct port of one R source file in
92
93
  the upstream package, function-for-function:
93
94
 
94
95
  | Python module | Ported from (R) | Notes |
@@ -120,10 +121,13 @@ beyond the R package's scope:
120
121
  | Python module | What it is |
121
122
  |---|---|
122
123
  | `_tables.py` | Name → `TwbParser`-accessor registry shared by `cli.py`, `webgui.py`, `diff.py`, and `batch.py`. Adding a new extractor to `TwbParser`? Add it here too so it's automatically available everywhere else. |
123
- | `cli.py` | `twbparser` command-line entry point, plus the `diff`/`batch` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
124
- | `webgui.py` | `twbparser-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
124
+ | `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename`/`template` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
125
+ | `webgui.py` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
125
126
  | `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
126
127
  | `diff.py` | `diff_tables()`/`diff_workbooks()`: row-level added/removed diff between two workbooks' same-named table, via `_tables.TABLE_SPECS`. No "changed" classification without a natural key — a changed row shows as one removed + one added row. |
128
+ | `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `suggest_renames()` (the same for every kind of object: field, parameter, worksheet, dashboard, datasource, folder, hierarchy; adds a `kind` column), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` and `report-renames` tables in `_tables.py`. A worksheet/dashboard rename must rewrite every reference (`_SHEET_REFERENCES`); if you learn of another place Tableau writes a sheet name, add it there. |
129
+ | `usage.py` | `field_usage()`: for every field, the worksheets, dashboards and calculations that use it, followed through calculations, groups/sets and bins (the `field-usage` table). Python-native rather than a port of the R package's field-usage helper. |
130
+ | `templates.py` | Workbook templates. `make_template()` writes a `.twbx` with a `template.json` manifest (required fields from `usage.py`, parameters, connections; data, extracts and credentials stripped); `read_data()` describes new data (CSV, or a `.twb`/`.twbx`/`.tds`); `suggest_mapping()` pairs fields with columns (reuses `rename._match_key`, type checks against the field's *physical* type, `datatype-customized` fields keep their type); `apply_template()` replaces the datasource's connection, keeping every field's local name, sets parameters and stores `template-answers.json`. A CSV connection is written in the 2020.2+ object-model shape (both `_.fcp.ObjectModelEncapsulateLegacy` relations, object graph, table column, `object-id` per record) when the template uses it -- keep those in step if you touch one. Untested in Tableau itself. |
127
131
  | `batch.py` | `scan_folder()`: runs one table across every `.twb`/`.twbx` in a directory, concatenated with a `workbook` column. Skips (with a warning) any file that fails to load/extract rather than aborting the batch. |
128
132
 
129
133
  ## Porting conventions (read before adding/modifying a function)
@@ -173,6 +177,12 @@ beyond the R package's scope:
173
177
  synthetic XML snippet (see `tests/test_joins.py`,
174
178
  `tests/test_dashboards.py` for the pattern — many are lifted from the R
175
179
  functions' own `@examples` roxygen blocks).
180
+ - `tests/corpus/` is 200 real workbooks (permissive licences, pinned by blob sha in
181
+ `manifest.csv`, licence texts in `licenses/`). The files are gitignored; fetch with
182
+ `python scripts/fetch_corpus.py`. `tests/test_corpus.py` skips without them. Run it when you
183
+ change anything that reads workbook XML: it found a Unicode matching bug the hand-made
184
+ fixtures could not. Add to the corpus only from repositories whose licence permits
185
+ redistribution, and record the licence text.
176
186
  - `tests/conftest.py` provides `wenjie_xml`, `wenjie_path`,
177
187
  `zip_twbx_path` fixtures and an `xml_from_string()` helper. Tests import
178
188
  it with `from conftest import xml_from_string` (no `tests/__init__.py`,
@@ -4,7 +4,7 @@ This project is a Python port of logic originally implemented in the R
4
4
  package "twbparser" (https://github.com/PrigasG/twbparser),
5
5
  Copyright (c) 2025 George Arthur.
6
6
 
7
- Copyright (c) 2026 the twbparser-py contributors
7
+ Copyright (c) 2026 the py-tbparse contributors
8
8
 
9
9
  Permission is hereby granted, free of charge, to any person obtaining a copy
10
10
  of this software and associated documentation files (the "Software"), to deal
@@ -0,0 +1,250 @@
1
+ Metadata-Version: 2.4
2
+ Name: py-tbparse
3
+ Version: 0.4.0
4
+ Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
5
+ Author: DDSNA
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/DDSNA/py-tbparse
8
+ Project-URL: Source, https://github.com/DDSNA/py-tbparse
9
+ Project-URL: Issues, https://github.com/DDSNA/py-tbparse/issues
10
+ Keywords: tableau,twb,twbx,workbook,parser,pandas
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
+ Classifier: Topic :: Office/Business
23
+ Requires-Python: >=3.9
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: lxml>=4.9
27
+ Requires-Dist: pandas>=1.5
28
+ Provides-Extra: test
29
+ Requires-Dist: pytest>=7; extra == "test"
30
+ Provides-Extra: browser
31
+ Requires-Dist: playwright>=1.40; extra == "browser"
32
+ Provides-Extra: dev
33
+ Requires-Dist: build>=1.0; extra == "dev"
34
+ Requires-Dist: twine>=5.0; extra == "dev"
35
+ Dynamic: license-file
36
+
37
+ # py-tbparse
38
+
39
+ Reads Tableau workbooks (`.twb` and `.twbx`) and gives you what's in them as pandas DataFrames: datasources, fields, calculated fields, joins, relationships, dashboards, custom SQL. It's plain Python. You don't need Tableau or R installed.
40
+
41
+ It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The browser GUI, the command-line tool, workbook diffing and folder scanning are new here.
42
+
43
+ ![py-tbparse GUI, overview of a loaded workbook](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-overview.png)
44
+
45
+ ## Install
46
+
47
+ ```bash
48
+ pip install py-tbparse
49
+ ```
50
+
51
+ Or from a checkout:
52
+
53
+ ```bash
54
+ git clone https://github.com/DDSNA/py-tbparse.git
55
+ cd py-tbparse
56
+ pip install -e .
57
+ ```
58
+
59
+ One name throughout: `pip install py-tbparse`, `import py_tbparse`, run `py-tbparse` or `py-tbparse-gui`. (Releases up to 0.2.0 imported `twbparser_py` and ran `twbparser` / `twbparser-gui`; those names are gone.)
60
+
61
+ ## Using it from Python
62
+
63
+ ```python
64
+ from py_tbparse import TwbParser
65
+
66
+ p = TwbParser("workbook.twb") # or a .twbx
67
+ p.get_overview() # counts of everything
68
+ p.get_calculated_fields() # each calculation and its formula
69
+ p.get_relationships() # how the tables connect
70
+ ```
71
+
72
+ Every getter returns a DataFrame. The rest are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql` and `get_published_refs`. `get_relationship_graph_dot()` returns the data model as Graphviz text, and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
73
+
74
+ Two helpers work across workbooks:
75
+
76
+ ```python
77
+ from py_tbparse import diff_workbooks, scan_folder
78
+
79
+ diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
80
+ scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
81
+ ```
82
+
83
+ ### Cleaning field names after a datasource switch
84
+
85
+ Pointing a workbook at a new datasource that only partly matches the old schema tends to leave ugly names: `ORDER_ID`, `orderId`, `Order ID (Orders1)`, `Order ID1`. `suggest_field_renames` proposes a clean name for each field. It only reports; it never edits the workbook.
86
+
87
+ ```python
88
+ from py_tbparse import TwbParser, suggest_field_renames
89
+
90
+ new = TwbParser("after_switch.twb")
91
+ old = TwbParser("before_switch.twb") # optional: the schema you want to match
92
+
93
+ suggest_field_renames(new, reference=old, only_changed=True)
94
+ # name current suggested reason score
95
+ # [ORDER_ID] ORDER_ID Order ID matches reference 1.0
96
+ # [Sales Amount (Orders1)] Sales Amount (Orders1) Sales Amount matches reference 1.0
97
+ # [orderDate] orderDate Order Date normalized NaN
98
+ ```
99
+
100
+ With a `reference` (a workbook, a fields table or a plain list of names), fields that match by name ignoring case, separators and Tableau's duplicate suffixes take the reference's exact spelling, and near-misses above `fuzzy_cutoff` (default 0.85) are matched too. Everything else is tidied by `normalize_name(name, style)`, where `style` is `title` (default), `snake`, `lower` or `keep`. Names that are already clean (`YTD Sales`, `iPhone Units`, `Country/Region`) are left alone. `1` and `(Table1)` suffixes are only dropped when the plain name exists in the same datasource (`Address Line 2` and `Q1` are never treated as duplicates), fuzzy matches never cross a different number, and two fields never get the same suggestion (the loser stays as it is, with `reason` set to `conflict`).
101
+
102
+ `p.get_field_renames()` does the same from a parser.
103
+
104
+ **Rename everything in the report, not just fields.** Pass `kinds` (or `--all` / `--kinds` on the command line) and worksheets, dashboards, datasources, parameters, folders and hierarchies are covered too:
105
+
106
+ ```python
107
+ from py_tbparse import TwbParser, suggest_renames, apply_field_renames
108
+
109
+ p = TwbParser("report.twb")
110
+ suggest_renames(p, only_changed=True) # kind, datasource, name, current, suggested, ...
111
+ suggest_renames(p, kinds=["worksheet", "dashboard"]) # just the sheets
112
+ apply_field_renames(p, kinds="all") # writes report_renamed.twb
113
+ ```
114
+
115
+ ```bash
116
+ py-tbparse rename report.twb --all --only-changed
117
+ py-tbparse rename report.twb --kinds worksheet,dashboard --write-workbook
118
+ ```
119
+
120
+ Each kind is renamed the way Tableau does it: fields, parameters and datasources get a caption (their internal names stay, so formulas and sheets keep working); a worksheet or dashboard is renamed in every place its name is written (the sheet, its window and thumbnail, the zones of dashboards that show it, actions, story points); a folder or hierarchy gets its new name. Worksheets and dashboards share one namespace, as they do in Tableau, so two of them never end up with the same name. A `reference` workbook lends its spelling to objects of the same kind. Datasources that Tableau named itself (`federated.0grg...`) and nobody captioned are left out. The `kind` column also appears in the CSV, so **Edit the suggestions yourself** works for sheets too. The `report-renames` table lists all of it, and the GUI's Field renames view has an "Everything in the report" switch. As with fields, none of this has been opened in Tableau itself.
121
+
122
+ **Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
123
+
124
+ ```bash
125
+ py-tbparse rename new.twb -r old.twb -f csv -o mapping.csv
126
+ py-tbparse rename new.twb --apply mapping.csv # writes new_renamed.twb; add --write-workbook PATH to choose the file
127
+ ```
128
+
129
+ Your edits are applied as written, including rows the tool had marked `conflict`; two rows giving the same name in one datasource are rejected. From Python this is `apply_field_renames(p, renames=load_rename_mapping("mapping.csv"))`.
130
+
131
+ **What will stay broken.** `py-tbparse rename new.twb -r old.twb --missing` (or `compare_field_schemas(new, old)`) lists the fields with no counterpart after the switch: `old only` fields that nothing in the new source matches, and `new only` fields nothing in the old workbook matches, each with the closest name on the other side as a hint. Sheets using an `old only` field stay red after Replace Data Source until you map or recreate it.
132
+
133
+ To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...)`) writes `<name>_renamed.twb` / `.twbx` next to the original. It sets each field's caption, which is how Tableau renames a field; the internal names that formulas and sheets use are not touched, and a `.twbx` keeps all its other contents. A field that only exists as a physical column (typical right after a datasource switch) gets a new minimal `<column>` element carrying the caption; that shape follows what Tableau writes but I have not opened such files in Tableau itself. It never modifies the original and refuses to overwrite an existing file unless you pass `overwrite=True`.
134
+
135
+ **When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
136
+
137
+ ### Templates
138
+
139
+ Turn a finished workbook into a template, then make new workbooks from it with other data. Every sheet, dashboard, calculation and format comes along; only the data changes. The idea comes from Tableau's Accelerators and Power BI's `.pbit` files.
140
+
141
+ ```bash
142
+ py-tbparse template make sales.twbx # writes sales.template.twbx
143
+ py-tbparse template show sales.template.twbx # the fields it needs, and its parameters
144
+ py-tbparse template apply sales.template.twbx --data q3.csv # suggested mapping + what would break; writes nothing
145
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping-out map.csv # save the mapping to edit
146
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping map.csv -p "Top N=10" --write
147
+ ```
148
+
149
+ ```python
150
+ from py_tbparse import make_template, load_template, read_data, suggest_mapping, apply_template
151
+
152
+ t = load_template(make_template("sales.twbx"))
153
+ data = read_data("q3.csv") # or a .twb / .twbx / .tds already connected to the new data
154
+ suggest_mapping(t, data) # field, required, used_by, mapped_to, status, ...
155
+ apply_template(t, data, params={"Top N": "10"}) # writes sales_q3.twbx
156
+ ```
157
+
158
+ **What a template is.** An ordinary `.twbx` (Tableau still opens it) with a `template.json` manifest inside. The manifest lists the fields the workbook takes from its data, marking a field `required` when a sheet uses it, directly or through calculations, groups and sets. It also lists the parameters and where the data came from. Extracts, packaged data and cached query results are left out (`--keep-data` keeps them as sample data), and user names and passwords are blanked.
159
+
160
+ **Mapping.** Each required field is matched to a column of the new data by name, ignoring case and separators (`ORDER_DATE` → `Order Date`), with close spellings accepted above `--cutoff`. Types are checked like Tableau's Accelerator mapper: a text column is never offered for a number or a date, while integer vs decimal and date vs date-time map with a warning. A field whose type the author changed in Tableau keeps that type, and Tableau converts the column. Before anything is written you see which sheets would break for each field left without a column; writing then needs `--allow-missing`. Edit the mapping as a CSV, as with renames.
161
+
162
+ **What gets written.** The template's connection is replaced by one to the new data (a CSV file, or the connection of the workbook / `.tds` you pass). Every field keeps the local name its sheets and formulas use; only the physical column behind it changes. Parameter values are set with `-p NAME=VALUE`, checked against the parameter's type and its list of allowed values. The output (`<template>_<data>.twbx` beside the template, never overwritten) also stores `template-answers.json`: which template, data, mapping and parameters made it, so it can be re-made or checked later.
163
+
164
+ **Limits.** A CSV feeds one table. A template whose datasource joins several tables needs a workbook or `.tds` as its data, so the joins come along. Excel files are not read directly yet; save as CSV or pass a workbook connected to the sheet. As with renames, I have not opened the generated workbooks in Tableau itself, so check one before relying on it.
165
+
166
+ `p.get_field_usage()` (or `field_usage(p)`, the `field-usage` table) is the analysis behind `required`: for every field, the sheets, dashboards and calculations that use it.
167
+
168
+ ### `.twbx` files
169
+
170
+ A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
171
+
172
+ ```python
173
+ from py_tbparse import extract_twb_from_twbx, twbx_extract_files
174
+
175
+ extract_twb_from_twbx("workbook.twbx", extract_dir="out/")
176
+ twbx_extract_files("workbook.twbx", exdir="out/") # everything in the package
177
+ ```
178
+
179
+ ## Command line
180
+
181
+ ```bash
182
+ py-tbparse workbook.twb # overview
183
+ py-tbparse workbook.twb tables # list the tables
184
+ py-tbparse workbook.twb calculated-fields
185
+ py-tbparse workbook.twb fields --format csv -o fields.csv
186
+ py-tbparse workbook.twbx dashboard-sheets --dashboard "Sales Overview"
187
+ py-tbparse workbook.twb validate # exit code 2 if it finds a problem
188
+ py-tbparse workbook.twb graph > model.dot # --include-inferred adds the guessed links
189
+ py-tbparse diff old.twb new.twb datasources
190
+ py-tbparse batch ./workbooks datasources
191
+ py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
192
+ py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
193
+ py-tbparse rename report.twb --all --write-workbook # sheets, dashboards, datasources, ... too
194
+ py-tbparse template make sales.twbx # see Templates above
195
+ py-tbparse template apply sales.template.twbx --data q3.csv --write
196
+ ```
197
+
198
+ Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`, `field-usage`, `field-renames`, `report-renames`.
199
+
200
+ `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--all`, `--kinds`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
201
+
202
+ ## GUI
203
+
204
+ ```bash
205
+ py-tbparse-gui workbook.twb # opens your browser with it loaded
206
+ py-tbparse-gui # starts empty, paste a path and hit Load
207
+ py-tbparse-gui --no-browser --port 8765 # server only, for a machine with no display
208
+ ```
209
+
210
+ It uses only the standard library, so there's nothing more to install.
211
+
212
+ The sidebar lists every table with its row count. Click a column header to sort, type in the filter box to narrow rows (`/` jumps to it), click a row to see a long formula or SQL statement in full. The tiles on the overview open their tables. The graph view lets you copy or download the DOT text. The Field renames view has the buttons for the feature above: pick a style, datasource and optional reference workbook, then **Create fixed workbook** saves `<name>_renamed` beside the original (and says so if that file already exists) or **Download fixed workbook** sends it to your browser without saving anything. Everything else exports as CSV. It has a dark theme and works in a narrow window.
213
+
214
+ ![py-tbparse GUI, Field renames view with suggested clean names](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-field-renames.png)
215
+
216
+ It's meant to run on your own machine for one person. It refuses requests that come from other websites, but there's no login, so don't put it on a shared network.
217
+
218
+ ## Tests
219
+
220
+ ```bash
221
+ pip install -e ".[test]"
222
+ pytest
223
+ ```
224
+
225
+ The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
226
+
227
+ `tests/corpus/` lists 200 more real workbooks from public repositories with MIT, Apache-2.0, ISC or CC0 licences, for integration tests and as examples (manifest, licence texts and where each file came from are in its README). The files themselves are not in git (about 26 MB): run `python scripts/fetch_corpus.py` to download them, checked against the manifest. `tests/test_corpus.py` then runs every feature over all of them; it skips when they are not fetched.
228
+
229
+ The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
230
+
231
+ ```bash
232
+ pip install -e ".[test,browser]"
233
+ playwright install chromium # add --with-deps if you have root
234
+ ./scripts/setup-browser-libs.sh # without root, this unpacks the system libraries locally
235
+ pytest tests/test_gui_browser.py
236
+ ```
237
+
238
+ ## Compared with the R package
239
+
240
+ The code follows the R original function by function, and the aim is the same output. Not everything is ported. Missing so far: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, the analytics helpers (calculation complexity, field usage, replication brief) and the Shiny inspector. The GUI covers some of what the inspector did.
241
+
242
+ Where the R version has a bug, this one doesn't copy it. That currently covers joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
243
+
244
+ ## Why a Tableau parser
245
+
246
+ A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python isn't hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
247
+
248
+ ## Credit
249
+
250
+ Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
@@ -0,0 +1,214 @@
1
+ # py-tbparse
2
+
3
+ Reads Tableau workbooks (`.twb` and `.twbx`) and gives you what's in them as pandas DataFrames: datasources, fields, calculated fields, joins, relationships, dashboards, custom SQL. It's plain Python. You don't need Tableau or R installed.
4
+
5
+ It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The browser GUI, the command-line tool, workbook diffing and folder scanning are new here.
6
+
7
+ ![py-tbparse GUI, overview of a loaded workbook](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-overview.png)
8
+
9
+ ## Install
10
+
11
+ ```bash
12
+ pip install py-tbparse
13
+ ```
14
+
15
+ Or from a checkout:
16
+
17
+ ```bash
18
+ git clone https://github.com/DDSNA/py-tbparse.git
19
+ cd py-tbparse
20
+ pip install -e .
21
+ ```
22
+
23
+ One name throughout: `pip install py-tbparse`, `import py_tbparse`, run `py-tbparse` or `py-tbparse-gui`. (Releases up to 0.2.0 imported `twbparser_py` and ran `twbparser` / `twbparser-gui`; those names are gone.)
24
+
25
+ ## Using it from Python
26
+
27
+ ```python
28
+ from py_tbparse import TwbParser
29
+
30
+ p = TwbParser("workbook.twb") # or a .twbx
31
+ p.get_overview() # counts of everything
32
+ p.get_calculated_fields() # each calculation and its formula
33
+ p.get_relationships() # how the tables connect
34
+ ```
35
+
36
+ Every getter returns a DataFrame. The rest are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql` and `get_published_refs`. `get_relationship_graph_dot()` returns the data model as Graphviz text, and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
37
+
38
+ Two helpers work across workbooks:
39
+
40
+ ```python
41
+ from py_tbparse import diff_workbooks, scan_folder
42
+
43
+ diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
44
+ scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
45
+ ```
46
+
47
+ ### Cleaning field names after a datasource switch
48
+
49
+ Pointing a workbook at a new datasource that only partly matches the old schema tends to leave ugly names: `ORDER_ID`, `orderId`, `Order ID (Orders1)`, `Order ID1`. `suggest_field_renames` proposes a clean name for each field. It only reports; it never edits the workbook.
50
+
51
+ ```python
52
+ from py_tbparse import TwbParser, suggest_field_renames
53
+
54
+ new = TwbParser("after_switch.twb")
55
+ old = TwbParser("before_switch.twb") # optional: the schema you want to match
56
+
57
+ suggest_field_renames(new, reference=old, only_changed=True)
58
+ # name current suggested reason score
59
+ # [ORDER_ID] ORDER_ID Order ID matches reference 1.0
60
+ # [Sales Amount (Orders1)] Sales Amount (Orders1) Sales Amount matches reference 1.0
61
+ # [orderDate] orderDate Order Date normalized NaN
62
+ ```
63
+
64
+ With a `reference` (a workbook, a fields table or a plain list of names), fields that match by name ignoring case, separators and Tableau's duplicate suffixes take the reference's exact spelling, and near-misses above `fuzzy_cutoff` (default 0.85) are matched too. Everything else is tidied by `normalize_name(name, style)`, where `style` is `title` (default), `snake`, `lower` or `keep`. Names that are already clean (`YTD Sales`, `iPhone Units`, `Country/Region`) are left alone. `1` and `(Table1)` suffixes are only dropped when the plain name exists in the same datasource (`Address Line 2` and `Q1` are never treated as duplicates), fuzzy matches never cross a different number, and two fields never get the same suggestion (the loser stays as it is, with `reason` set to `conflict`).
65
+
66
+ `p.get_field_renames()` does the same from a parser.
67
+
68
+ **Rename everything in the report, not just fields.** Pass `kinds` (or `--all` / `--kinds` on the command line) and worksheets, dashboards, datasources, parameters, folders and hierarchies are covered too:
69
+
70
+ ```python
71
+ from py_tbparse import TwbParser, suggest_renames, apply_field_renames
72
+
73
+ p = TwbParser("report.twb")
74
+ suggest_renames(p, only_changed=True) # kind, datasource, name, current, suggested, ...
75
+ suggest_renames(p, kinds=["worksheet", "dashboard"]) # just the sheets
76
+ apply_field_renames(p, kinds="all") # writes report_renamed.twb
77
+ ```
78
+
79
+ ```bash
80
+ py-tbparse rename report.twb --all --only-changed
81
+ py-tbparse rename report.twb --kinds worksheet,dashboard --write-workbook
82
+ ```
83
+
84
+ Each kind is renamed the way Tableau does it: fields, parameters and datasources get a caption (their internal names stay, so formulas and sheets keep working); a worksheet or dashboard is renamed in every place its name is written (the sheet, its window and thumbnail, the zones of dashboards that show it, actions, story points); a folder or hierarchy gets its new name. Worksheets and dashboards share one namespace, as they do in Tableau, so two of them never end up with the same name. A `reference` workbook lends its spelling to objects of the same kind. Datasources that Tableau named itself (`federated.0grg...`) and nobody captioned are left out. The `kind` column also appears in the CSV, so **Edit the suggestions yourself** works for sheets too. The `report-renames` table lists all of it, and the GUI's Field renames view has an "Everything in the report" switch. As with fields, none of this has been opened in Tableau itself.
85
+
86
+ **Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
87
+
88
+ ```bash
89
+ py-tbparse rename new.twb -r old.twb -f csv -o mapping.csv
90
+ py-tbparse rename new.twb --apply mapping.csv # writes new_renamed.twb; add --write-workbook PATH to choose the file
91
+ ```
92
+
93
+ Your edits are applied as written, including rows the tool had marked `conflict`; two rows giving the same name in one datasource are rejected. From Python this is `apply_field_renames(p, renames=load_rename_mapping("mapping.csv"))`.
94
+
95
+ **What will stay broken.** `py-tbparse rename new.twb -r old.twb --missing` (or `compare_field_schemas(new, old)`) lists the fields with no counterpart after the switch: `old only` fields that nothing in the new source matches, and `new only` fields nothing in the old workbook matches, each with the closest name on the other side as a hint. Sheets using an `old only` field stay red after Replace Data Source until you map or recreate it.
96
+
97
+ To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...)`) writes `<name>_renamed.twb` / `.twbx` next to the original. It sets each field's caption, which is how Tableau renames a field; the internal names that formulas and sheets use are not touched, and a `.twbx` keeps all its other contents. A field that only exists as a physical column (typical right after a datasource switch) gets a new minimal `<column>` element carrying the caption; that shape follows what Tableau writes but I have not opened such files in Tableau itself. It never modifies the original and refuses to overwrite an existing file unless you pass `overwrite=True`.
98
+
99
+ **When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
100
+
101
+ ### Templates
102
+
103
+ Turn a finished workbook into a template, then make new workbooks from it with other data. Every sheet, dashboard, calculation and format comes along; only the data changes. The idea comes from Tableau's Accelerators and Power BI's `.pbit` files.
104
+
105
+ ```bash
106
+ py-tbparse template make sales.twbx # writes sales.template.twbx
107
+ py-tbparse template show sales.template.twbx # the fields it needs, and its parameters
108
+ py-tbparse template apply sales.template.twbx --data q3.csv # suggested mapping + what would break; writes nothing
109
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping-out map.csv # save the mapping to edit
110
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping map.csv -p "Top N=10" --write
111
+ ```
112
+
113
+ ```python
114
+ from py_tbparse import make_template, load_template, read_data, suggest_mapping, apply_template
115
+
116
+ t = load_template(make_template("sales.twbx"))
117
+ data = read_data("q3.csv") # or a .twb / .twbx / .tds already connected to the new data
118
+ suggest_mapping(t, data) # field, required, used_by, mapped_to, status, ...
119
+ apply_template(t, data, params={"Top N": "10"}) # writes sales_q3.twbx
120
+ ```
121
+
122
+ **What a template is.** An ordinary `.twbx` (Tableau still opens it) with a `template.json` manifest inside. The manifest lists the fields the workbook takes from its data, marking a field `required` when a sheet uses it, directly or through calculations, groups and sets. It also lists the parameters and where the data came from. Extracts, packaged data and cached query results are left out (`--keep-data` keeps them as sample data), and user names and passwords are blanked.
123
+
124
+ **Mapping.** Each required field is matched to a column of the new data by name, ignoring case and separators (`ORDER_DATE` → `Order Date`), with close spellings accepted above `--cutoff`. Types are checked like Tableau's Accelerator mapper: a text column is never offered for a number or a date, while integer vs decimal and date vs date-time map with a warning. A field whose type the author changed in Tableau keeps that type, and Tableau converts the column. Before anything is written you see which sheets would break for each field left without a column; writing then needs `--allow-missing`. Edit the mapping as a CSV, as with renames.
125
+
126
+ **What gets written.** The template's connection is replaced by one to the new data (a CSV file, or the connection of the workbook / `.tds` you pass). Every field keeps the local name its sheets and formulas use; only the physical column behind it changes. Parameter values are set with `-p NAME=VALUE`, checked against the parameter's type and its list of allowed values. The output (`<template>_<data>.twbx` beside the template, never overwritten) also stores `template-answers.json`: which template, data, mapping and parameters made it, so it can be re-made or checked later.
127
+
128
+ **Limits.** A CSV feeds one table. A template whose datasource joins several tables needs a workbook or `.tds` as its data, so the joins come along. Excel files are not read directly yet; save as CSV or pass a workbook connected to the sheet. As with renames, I have not opened the generated workbooks in Tableau itself, so check one before relying on it.
129
+
130
+ `p.get_field_usage()` (or `field_usage(p)`, the `field-usage` table) is the analysis behind `required`: for every field, the sheets, dashboards and calculations that use it.
131
+
132
+ ### `.twbx` files
133
+
134
+ A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
135
+
136
+ ```python
137
+ from py_tbparse import extract_twb_from_twbx, twbx_extract_files
138
+
139
+ extract_twb_from_twbx("workbook.twbx", extract_dir="out/")
140
+ twbx_extract_files("workbook.twbx", exdir="out/") # everything in the package
141
+ ```
142
+
143
+ ## Command line
144
+
145
+ ```bash
146
+ py-tbparse workbook.twb # overview
147
+ py-tbparse workbook.twb tables # list the tables
148
+ py-tbparse workbook.twb calculated-fields
149
+ py-tbparse workbook.twb fields --format csv -o fields.csv
150
+ py-tbparse workbook.twbx dashboard-sheets --dashboard "Sales Overview"
151
+ py-tbparse workbook.twb validate # exit code 2 if it finds a problem
152
+ py-tbparse workbook.twb graph > model.dot # --include-inferred adds the guessed links
153
+ py-tbparse diff old.twb new.twb datasources
154
+ py-tbparse batch ./workbooks datasources
155
+ py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
156
+ py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
157
+ py-tbparse rename report.twb --all --write-workbook # sheets, dashboards, datasources, ... too
158
+ py-tbparse template make sales.twbx # see Templates above
159
+ py-tbparse template apply sales.template.twbx --data q3.csv --write
160
+ ```
161
+
162
+ Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`, `field-usage`, `field-renames`, `report-renames`.
163
+
164
+ `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--all`, `--kinds`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
165
+
166
+ ## GUI
167
+
168
+ ```bash
169
+ py-tbparse-gui workbook.twb # opens your browser with it loaded
170
+ py-tbparse-gui # starts empty, paste a path and hit Load
171
+ py-tbparse-gui --no-browser --port 8765 # server only, for a machine with no display
172
+ ```
173
+
174
+ It uses only the standard library, so there's nothing more to install.
175
+
176
+ The sidebar lists every table with its row count. Click a column header to sort, type in the filter box to narrow rows (`/` jumps to it), click a row to see a long formula or SQL statement in full. The tiles on the overview open their tables. The graph view lets you copy or download the DOT text. The Field renames view has the buttons for the feature above: pick a style, datasource and optional reference workbook, then **Create fixed workbook** saves `<name>_renamed` beside the original (and says so if that file already exists) or **Download fixed workbook** sends it to your browser without saving anything. Everything else exports as CSV. It has a dark theme and works in a narrow window.
177
+
178
+ ![py-tbparse GUI, Field renames view with suggested clean names](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-field-renames.png)
179
+
180
+ It's meant to run on your own machine for one person. It refuses requests that come from other websites, but there's no login, so don't put it on a shared network.
181
+
182
+ ## Tests
183
+
184
+ ```bash
185
+ pip install -e ".[test]"
186
+ pytest
187
+ ```
188
+
189
+ The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
190
+
191
+ `tests/corpus/` lists 200 more real workbooks from public repositories with MIT, Apache-2.0, ISC or CC0 licences, for integration tests and as examples (manifest, licence texts and where each file came from are in its README). The files themselves are not in git (about 26 MB): run `python scripts/fetch_corpus.py` to download them, checked against the manifest. `tests/test_corpus.py` then runs every feature over all of them; it skips when they are not fetched.
192
+
193
+ The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
194
+
195
+ ```bash
196
+ pip install -e ".[test,browser]"
197
+ playwright install chromium # add --with-deps if you have root
198
+ ./scripts/setup-browser-libs.sh # without root, this unpacks the system libraries locally
199
+ pytest tests/test_gui_browser.py
200
+ ```
201
+
202
+ ## Compared with the R package
203
+
204
+ The code follows the R original function by function, and the aim is the same output. Not everything is ported. Missing so far: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, the analytics helpers (calculation complexity, field usage, replication brief) and the Shiny inspector. The GUI covers some of what the inspector did.
205
+
206
+ Where the R version has a bug, this one doesn't copy it. That currently covers joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
207
+
208
+ ## Why a Tableau parser
209
+
210
+ A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python isn't hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
211
+
212
+ ## Credit
213
+
214
+ Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
@@ -1,4 +1,4 @@
1
- """twbparser_py: a native Python port of the twbparser R package.
1
+ """py_tbparse: a native Python port of the twbparser R package.
2
2
 
3
3
  Parses Tableau .twb/.twbx workbook files into pandas DataFrames. Ported
4
4
  from https://github.com/PrigasG/twbparser (MIT licensed).
@@ -15,7 +15,26 @@ from .joins import extract_joins
15
15
  from .parser import TwbParser
16
16
  from .published import extract_published_refs
17
17
  from .relationships import extract_relations, extract_relationships
18
+ from .rename import (
19
+ apply_field_renames,
20
+ compare_field_schemas,
21
+ load_rename_mapping,
22
+ normalize_name,
23
+ suggest_field_renames,
24
+ suggest_renames,
25
+ )
18
26
  from .sql import extract_custom_sql, extract_initial_sql
27
+ from .templates import (
28
+ Template,
29
+ TemplateError,
30
+ apply_template,
31
+ load_mapping,
32
+ load_template,
33
+ make_template,
34
+ read_data,
35
+ suggest_mapping,
36
+ )
37
+ from .usage import field_usage
19
38
  from .validators import validate_relationships
20
39
  from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
21
40
 
@@ -44,6 +63,21 @@ __all__ = [
44
63
  "diff_tables",
45
64
  "diff_workbooks",
46
65
  "scan_folder",
66
+ "apply_field_renames",
67
+ "compare_field_schemas",
68
+ "load_rename_mapping",
69
+ "normalize_name",
70
+ "suggest_field_renames",
71
+ "suggest_renames",
72
+ "field_usage",
73
+ "Template",
74
+ "TemplateError",
75
+ "make_template",
76
+ "load_template",
77
+ "read_data",
78
+ "suggest_mapping",
79
+ "load_mapping",
80
+ "apply_template",
47
81
  ]
48
82
 
49
83
  try: