py-tbparse 0.3.0__tar.gz → 0.4.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/AGENTS.md +72 -16
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/MANIFEST.in +1 -0
- py_tbparse-0.4.4/PKG-INFO +137 -0
- py_tbparse-0.4.4/README.md +101 -0
- py_tbparse-0.4.4/docs/gui-field-renames.png +0 -0
- py_tbparse-0.4.4/docs/gui-graph.png +0 -0
- py_tbparse-0.4.4/docs/gui-overview.png +0 -0
- py_tbparse-0.4.4/docs/gui-themes.png +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/__init__.py +24 -1
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/_tables.py +24 -1
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/cli.py +124 -1
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/graph.py +51 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/parser.py +37 -1
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/rename.py +272 -33
- py_tbparse-0.4.4/py_tbparse/report.py +124 -0
- py_tbparse-0.4.4/py_tbparse/templates.py +855 -0
- py_tbparse-0.4.4/py_tbparse/usage.py +248 -0
- py_tbparse-0.4.4/py_tbparse/webgui.py +657 -0
- py_tbparse-0.4.4/py_tbparse/webui/app.css +318 -0
- py_tbparse-0.4.4/py_tbparse/webui/app.js +1497 -0
- py_tbparse-0.4.4/py_tbparse/webui/graph.js +453 -0
- py_tbparse-0.4.4/py_tbparse/webui/index.html +138 -0
- py_tbparse-0.4.4/py_tbparse/webui/table.js +359 -0
- py_tbparse-0.4.4/py_tbparse/webui/themes.css +237 -0
- py_tbparse-0.4.4/py_tbparse/webui/tokens.css +104 -0
- py_tbparse-0.4.4/py_tbparse.egg-info/PKG-INFO +137 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/SOURCES.txt +26 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/pyproject.toml +2 -2
- py_tbparse-0.4.4/tests/conftest.py +160 -0
- py_tbparse-0.4.4/tests/test_corpus.py +84 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_graph.py +27 -0
- py_tbparse-0.4.4/tests/test_gui_a11y.py +476 -0
- py_tbparse-0.4.4/tests/test_gui_browser.py +1068 -0
- py_tbparse-0.4.4/tests/test_gui_graph.py +401 -0
- py_tbparse-0.4.4/tests/test_gui_layout.py +304 -0
- py_tbparse-0.4.4/tests/test_gui_open.py +167 -0
- py_tbparse-0.4.4/tests/test_gui_overview.py +167 -0
- py_tbparse-0.4.4/tests/test_gui_table.py +1111 -0
- py_tbparse-0.4.4/tests/test_gui_themes.py +229 -0
- py_tbparse-0.4.4/tests/test_overview_endpoint.py +81 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_public_workbooks.py +18 -0
- py_tbparse-0.4.4/tests/test_rename_all.py +215 -0
- py_tbparse-0.4.4/tests/test_report.py +214 -0
- py_tbparse-0.4.4/tests/test_templates.py +415 -0
- py_tbparse-0.4.4/tests/test_upload.py +189 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_webgui.py +108 -18
- py_tbparse-0.4.4/tests/test_webui_tokens.py +175 -0
- py_tbparse-0.3.0/PKG-INFO +0 -196
- py_tbparse-0.3.0/README.md +0 -160
- py_tbparse-0.3.0/docs/gui-field-renames.png +0 -0
- py_tbparse-0.3.0/docs/gui-overview.png +0 -0
- py_tbparse-0.3.0/py_tbparse/webgui.py +0 -1111
- py_tbparse-0.3.0/py_tbparse.egg-info/PKG-INFO +0 -196
- py_tbparse-0.3.0/tests/conftest.py +0 -56
- py_tbparse-0.3.0/tests/test_gui_browser.py +0 -452
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/LICENSE +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/_clean.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/_xml.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/batch.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/calculated_fields.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/dashboards.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/datasources.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/diff.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/fields.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/joins.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/published.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/py.typed +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/relationships.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/sql.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/validators.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/dependency_links.txt +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/entry_points.txt +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/requires.txt +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/top_level.txt +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/setup.cfg +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/fixtures/test_for_wenjie.twb +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/fixtures/test_for_zip.twbx +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_batch.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_calculated_fields.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_clean.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_cli.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_dashboards.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_datasources.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_diff.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_fields.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_joins.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_parser.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_published.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_relationships.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_rename.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_rename_mapping.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_sql.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_validators.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_version.py +0 -0
- {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_xml.py +0 -0
|
@@ -14,14 +14,15 @@ type="join">` and 2020.2+ `<relationships>`), inferred relationships,
|
|
|
14
14
|
dashboards/dashboard-sheets, `validate_relationships`, custom/initial SQL,
|
|
15
15
|
and published-source detection. **Not yet ported** (v2): formatting,
|
|
16
16
|
tooltips, colors, axes, sorts, dashboard layout/actions, analytics
|
|
17
|
-
helpers (calc complexity,
|
|
17
|
+
helpers (calc complexity, replication brief), and the
|
|
18
18
|
Shiny-inspector equivalent.
|
|
19
19
|
|
|
20
20
|
Also added, with no R equivalent (Python-native extras -- see "Non-R
|
|
21
21
|
modules" below): Graphviz DOT export of the relationship graph
|
|
22
22
|
(replaces the R package's igraph/ggraph-based plotting with a
|
|
23
23
|
dependency-free alternative), workbook-to-workbook diff, folder/batch
|
|
24
|
-
analysis across many workbooks,
|
|
24
|
+
analysis across many workbooks, field renaming, field usage, workbook
|
|
25
|
+
templates, and Jupyter rich display.
|
|
25
26
|
|
|
26
27
|
## Setup
|
|
27
28
|
|
|
@@ -120,11 +121,14 @@ beyond the R package's scope:
|
|
|
120
121
|
| Python module | What it is |
|
|
121
122
|
|---|---|
|
|
122
123
|
| `_tables.py` | Name → `TwbParser`-accessor registry shared by `cli.py`, `webgui.py`, `diff.py`, and `batch.py`. Adding a new extractor to `TwbParser`? Add it here too so it's automatically available everywhere else. |
|
|
123
|
-
| `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
|
|
124
|
-
| `webgui.py` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
|
|
125
|
-
| `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
|
|
124
|
+
| `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename`/`template` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
|
|
125
|
+
| `webgui.py` + `webui/` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. The server is `webgui.py`; the page is `webui/` (`index.html`, `tokens.css`, `themes.css`, `app.css`, `table.js`, `graph.js`, `app.js`). Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
|
|
126
|
+
| `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges); `graph_data()` is the same edges as plain data, which the GUI lays out and draws itself (`webui/graph.js`: layered layout, SVG, pan/zoom, keyboard). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
|
|
126
127
|
| `diff.py` | `diff_tables()`/`diff_workbooks()`: row-level added/removed diff between two workbooks' same-named table, via `_tables.TABLE_SPECS`. No "changed" classification without a natural key — a changed row shows as one removed + one added row. |
|
|
127
|
-
| `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames`
|
|
128
|
+
| `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `suggest_renames()` (the same for every kind of object: field, parameter, worksheet, dashboard, datasource, folder, hierarchy; adds a `kind` column), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` and `report-renames` tables in `_tables.py`. A worksheet/dashboard rename must rewrite every reference (`_SHEET_REFERENCES`); if you learn of another place Tableau writes a sheet name, add it there. |
|
|
129
|
+
| `report.py` | `workbook_report()`: the overview's report card (summary sentence, health checks, dashboards with their sheets) built from `validate_relationships`, `field_usage` and `usage.missing_references`. Every health item names the table and column filters that list exactly the rows it counted; a test keeps count and rows equal on all 200 corpus workbooks. |
|
|
130
|
+
| `usage.py` | `field_usage()`: for every field, the worksheets, dashboards and calculations that use it, followed through calculations, groups/sets and bins (the `field-usage` table). Python-native rather than a port of the R package's field-usage helper. |
|
|
131
|
+
| `templates.py` | Workbook templates. `make_template()` writes a `.twbx` with a `template.json` manifest (required fields from `usage.py`, parameters, connections; data, extracts and credentials stripped); `read_data()` describes new data (CSV, or a `.twb`/`.twbx`/`.tds`); `suggest_mapping()` pairs fields with columns (reuses `rename._match_key`, type checks against the field's *physical* type, `datatype-customized` fields keep their type); `apply_template()` replaces the datasource's connection, keeping every field's local name, sets parameters and stores `template-answers.json`. A CSV connection is written in the 2020.2+ object-model shape (both `_.fcp.ObjectModelEncapsulateLegacy` relations, object graph, table column, `object-id` per record) when the template uses it -- keep those in step if you touch one. Untested in Tableau itself. |
|
|
128
132
|
| `batch.py` | `scan_folder()`: runs one table across every `.twb`/`.twbx` in a directory, concatenated with a `workbook` column. Skips (with a warning) any file that fails to load/extract rather than aborting the batch. |
|
|
129
133
|
|
|
130
134
|
## Porting conventions (read before adding/modifying a function)
|
|
@@ -174,6 +178,12 @@ beyond the R package's scope:
|
|
|
174
178
|
synthetic XML snippet (see `tests/test_joins.py`,
|
|
175
179
|
`tests/test_dashboards.py` for the pattern — many are lifted from the R
|
|
176
180
|
functions' own `@examples` roxygen blocks).
|
|
181
|
+
- `tests/corpus/` is 200 real workbooks (permissive licences, pinned by blob sha in
|
|
182
|
+
`manifest.csv`, licence texts in `licenses/`). The files are gitignored; fetch with
|
|
183
|
+
`python scripts/fetch_corpus.py`. `tests/test_corpus.py` skips without them. Run it when you
|
|
184
|
+
change anything that reads workbook XML: it found a Unicode matching bug the hand-made
|
|
185
|
+
fixtures could not. Add to the corpus only from repositories whose licence permits
|
|
186
|
+
redistribution, and record the licence text.
|
|
177
187
|
- `tests/conftest.py` provides `wenjie_xml`, `wenjie_path`,
|
|
178
188
|
`zip_twbx_path` fixtures and an `xml_from_string()` helper. Tests import
|
|
179
189
|
it with `from conftest import xml_from_string` (no `tests/__init__.py`,
|
|
@@ -194,21 +204,38 @@ string became a literal newline, splitting a JS string literal across
|
|
|
194
204
|
two lines) killed the entire script, so no handlers bound and the UI was
|
|
195
205
|
inert — with the whole suite green.
|
|
196
206
|
|
|
197
|
-
So: **any change to
|
|
207
|
+
So: **any change to the page (`py_tbparse/webui/`) needs a browser test**, in
|
|
208
|
+
The browser suite is split by concern, all sharing the fixtures in `tests/test_gui_browser.py`:
|
|
209
|
+
`test_gui_table.py` (windowing invariants, pipeline vs. a Python oracle, a seeded random walk;
|
|
210
|
+
`PYTBPARSE_WALK_SEEDS`/`PYTBPARSE_WALK_STEPS` widen it), `test_gui_a11y.py` (roles, keyboard, focus, rendered
|
|
211
|
+
contrast in both themes) and `test_gui_layout.py` (no sideways overflow from 320 px up, layout stability).
|
|
212
|
+
Design decisions the tests pin: only sorting uses a view transition (Chromium sends clicks to the page root while
|
|
213
|
+
one runs); column `MIN_WIDTH` is 80; per-table view settings are keyed by column name; disabled menu items use
|
|
214
|
+
`aria-disabled` so they stay focusable.
|
|
215
|
+
|
|
198
216
|
`tests/test_gui_browser.py`, which runs the page in real headless
|
|
199
217
|
Chromium via Playwright and fails on any uncaught JS error. The cheap
|
|
200
218
|
structural guards in `test_webgui.py` (unterminated string literals,
|
|
201
219
|
bracket balance) are a backstop, not a substitute.
|
|
202
220
|
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
(
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
the
|
|
221
|
+
The page is real files, not a Python string: `py_tbparse/webui/index.html`, `tokens.css` (the design
|
|
222
|
+
tokens), `app.css` and `app.js`. They are served by `webgui.py` from a fixed whitelist under
|
|
223
|
+
`/static/` (`_ASSETS`), so a request can never reach any other file; add a new asset to
|
|
224
|
+
`_ASSETS` and to the `webui/*` package-data (pyproject and MANIFEST.in) or it will not ship.
|
|
225
|
+
`index.html` is the only thing that gets server values: `_render_index()` replaces
|
|
226
|
+
`<!--APP_CONFIG-->` (now in `<head>`, so the saved theme is on `<html>` before the first paint) with the single
|
|
227
|
+
inline `<script>`. Keep it that way (one inline script plus the three external ones, `table.js`, `graph.js`, then
|
|
228
|
+
`app.js`); a test counts them. `populateTables();` must appear exactly once
|
|
229
|
+
in `app.js`, and the structural tests (unterminated string literals, bracket balance) read both served scripts.
|
|
230
|
+
Keep apostrophes out of JS strings and comments (the test counts quotes per line).
|
|
231
|
+
|
|
232
|
+
`POST /upload` (drag and drop, Open file) takes raw bytes with `Content-Type: application/octet-stream` and the name
|
|
233
|
+
in `X-Filename`: neither is CORS-safelisted, so another site cannot send it without a preflight. It streams to a
|
|
234
|
+
`mkdtemp` directory, refuses over `MAX_UPLOAD_BYTES` (413) and content that does not match the extension, and keeps
|
|
235
|
+
only one upload at a time. An uploaded workbook cannot "create beside the original" (409), only download.
|
|
236
|
+
|
|
237
|
+
Any server-side value spliced into the page (currently `TABLE_NAMES`, the preloaded workbook path
|
|
238
|
+
and the version, all in `_render_index()`'s config script) must go through `_json_for_script()`, not
|
|
212
239
|
bare `json.dumps()`. `json.dumps` doesn't escape `/`, so a value
|
|
213
240
|
containing the literal text `</script>` closes the script tag early in
|
|
214
241
|
the browser's HTML parser — this is real, not theoretical, since the
|
|
@@ -232,6 +259,35 @@ leaves inputs empty and tests time out) and may crash rendering. Don't add
|
|
|
232
259
|
startup, which makes collection fail for anyone who doesn't have it
|
|
233
260
|
installed.
|
|
234
261
|
|
|
262
|
+
The table is windowed (`table.js`, class `VTable`): rows have a fixed height (`--row-h`) and only the rows
|
|
263
|
+
near the viewport exist in the page, so code and tests that read rows from the DOM see a window, not the table.
|
|
264
|
+
Use `aria-rowcount` for the size and the column menu's "Copy column values" to read a whole column (the tests do:
|
|
265
|
+
`_copy_column`). Never render every row; the old table needed over 4 s to sort 20,000 rows and could not draw
|
|
266
|
+
50,000. Data work (search index, typed sort) is in `app.js`; each render records the User Timing measure
|
|
267
|
+
`py-tbparse:table`, which the speed tests read. The budgets (first paint of 1,000 rows 150 ms, filtering 50,000 rows
|
|
268
|
+
100 ms) are enforced at twice the figure. Heavy tests skip below 1.5 GB of free memory (`_enough_memory`); this
|
|
269
|
+
sandbox has only 2 CPUs and 7.9 GB, and a 50,000-row DOM render has crashed it before.
|
|
270
|
+
|
|
271
|
+
The page's colours, spacing, type and motion are tokens in `webui/tokens.css`; use them rather than literals.
|
|
272
|
+
Colour themes are `webui/themes.css`: each theme is a complete colour set in light and dark, selected by
|
|
273
|
+
`<html data-theme data-mode>` (Auto is resolved to light/dark by the head script and kept in step by `app.js`;
|
|
274
|
+
dark colours live under `[data-mode="dark"]`, there is no `prefers-color-scheme` query). To add a theme, add its
|
|
275
|
+
two blocks there, its id to `webgui.THEMES` and its label to `THEME_LABELS` in `app.js`; the token test checks it.
|
|
276
|
+
`tests/test_webui_tokens.py` reads those files and fails if any text/background pair drops below WCAG AA
|
|
277
|
+
4.5:1, if the stylesheet uses an undefined variable, or if text is faded with `opacity` (use `--faint`).
|
|
278
|
+
Motion durations come from the `--dur-*` tokens, which `prefers-reduced-motion` sets to instant; keep new
|
|
279
|
+
animations on those tokens.
|
|
280
|
+
|
|
281
|
+
`scripts/readme_screenshots.py` retakes the README's four images from the made-up `docs/demo/coffee-shop.twb`
|
|
282
|
+
(built by `scripts/make_demo_workbook.py`, which is the one place that workbook is defined); run both after a
|
|
283
|
+
visible change.
|
|
284
|
+
|
|
285
|
+
`scripts/gui_screenshots.py WORKBOOK OUT_DIR [--compare BASELINE_DIR]` captures nine GUI states (start,
|
|
286
|
+
overview and fields in light and dark, renames, graph, phone width) deterministically. Use it for GUI
|
|
287
|
+
refactors: a pure refactor must compare all-identical to the baseline taken before it; a redesign is expected to
|
|
288
|
+
differ, so review the new look and re-capture the baseline. The redesign plan and its decisions are in
|
|
289
|
+
`docs/ui-redesign-plan.md`.
|
|
290
|
+
|
|
235
291
|
## Commit / PR conventions
|
|
236
292
|
|
|
237
293
|
Nothing project-specific beyond the harness defaults — see repo commit
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: py-tbparse
|
|
3
|
+
Version: 0.4.4
|
|
4
|
+
Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
|
|
5
|
+
Author: DDSNA
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/DDSNA/py-tbparse
|
|
8
|
+
Project-URL: Source, https://github.com/DDSNA/py-tbparse
|
|
9
|
+
Project-URL: Issues, https://github.com/DDSNA/py-tbparse/issues
|
|
10
|
+
Keywords: tableau,twb,twbx,workbook,parser,pandas
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
+
Classifier: Topic :: Office/Business
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: lxml>=4.9
|
|
27
|
+
Requires-Dist: pandas>=1.5
|
|
28
|
+
Provides-Extra: test
|
|
29
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
30
|
+
Provides-Extra: browser
|
|
31
|
+
Requires-Dist: playwright>=1.40; extra == "browser"
|
|
32
|
+
Provides-Extra: dev
|
|
33
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
34
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# py-tbparse
|
|
38
|
+
|
|
39
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
40
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
41
|
+
[](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml)
|
|
42
|
+
[](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE)
|
|
43
|
+
|
|
44
|
+
Read Tableau workbooks (`.twb` and `.twbx`) from Python. You get datasources, fields, calculated fields, joins, relationships and dashboards as pandas DataFrames. It is plain Python on top of `lxml` and `pandas`, so Tableau and R are not needed.
|
|
45
|
+
|
|
46
|
+
It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The command-line tool, the browser GUI, renaming, templates, workbook diffing and folder scanning are new here.
|
|
47
|
+
|
|
48
|
+

|
|
49
|
+
|
|
50
|
+
## What it does
|
|
51
|
+
|
|
52
|
+
- Lists what is in a workbook: datasources, parameters, fields, calculations, joins, relationships, dashboards, custom SQL, published sources.
|
|
53
|
+
- Checks relationships and finds calculations that refer to fields the workbook does not have.
|
|
54
|
+
- Shows which worksheets, dashboards and calculations use each field.
|
|
55
|
+
- Suggests clean names for fields, sheets, dashboards and other objects, and writes a renamed copy.
|
|
56
|
+
- Turns a finished workbook into a template you can fill with other data.
|
|
57
|
+
- Compares two workbooks, scans a folder of them, and draws the data model as a graph.
|
|
58
|
+
- Works from Python, from the command line, or in a local browser page.
|
|
59
|
+
|
|
60
|
+
## Install
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
pip install py-tbparse
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Python 3.9 or newer. The package, the import and both commands all use the same name: `pip install py-tbparse`, `import py_tbparse`, `py-tbparse`, `py-tbparse-gui`. Releases up to 0.2.0 used `twbparser_py` and `twbparser`; those names are gone.
|
|
67
|
+
|
|
68
|
+
## Quick start
|
|
69
|
+
|
|
70
|
+
From Python:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from py_tbparse import TwbParser
|
|
74
|
+
|
|
75
|
+
p = TwbParser("workbook.twb") # or a .twbx
|
|
76
|
+
p.get_overview() # counts of everything
|
|
77
|
+
p.get_calculated_fields() # each calculation and its formula
|
|
78
|
+
p.get_relationships() # how the tables connect
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Every getter returns a DataFrame. The others are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql`, `get_published_refs` and `get_field_usage`. `get_relationship_graph_dot()` returns the data model as Graphviz text and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
|
|
82
|
+
|
|
83
|
+
From the command line:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
py-tbparse workbook.twb # overview
|
|
87
|
+
py-tbparse workbook.twb calculated-fields
|
|
88
|
+
py-tbparse workbook.twb fields --format csv -o fields.csv
|
|
89
|
+
py-tbparse workbook.twb validate # exit code 2 if it finds a problem
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
In the browser:
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
py-tbparse-gui workbook.twb
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Across workbooks:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
from py_tbparse import TwbParser, diff_workbooks, scan_folder
|
|
102
|
+
|
|
103
|
+
diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
|
|
104
|
+
scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## Guides
|
|
108
|
+
|
|
109
|
+
- [Command line](https://github.com/DDSNA/py-tbparse/blob/main/docs/cli.md): every table, the `diff`, `batch`, `rename` and `template` commands, and their options.
|
|
110
|
+
- [Browser GUI](https://github.com/DDSNA/py-tbparse/blob/main/docs/gui.md): opening files, the table, the overview, the graph, themes.
|
|
111
|
+
- [Renaming](https://github.com/DDSNA/py-tbparse/blob/main/docs/renaming.md): clean names after a datasource switch, editing the suggestions, what will stay broken.
|
|
112
|
+
- [Templates](https://github.com/DDSNA/py-tbparse/blob/main/docs/templates.md): make a template from a workbook and apply it to new data.
|
|
113
|
+
- [`.twbx` files](https://github.com/DDSNA/py-tbparse/blob/main/docs/twbx.md): they are read straight from the zip; how to extract the contents.
|
|
114
|
+
- [Development](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md): running the tests, the workbook corpus, the browser tests.
|
|
115
|
+
|
|
116
|
+
## Limits
|
|
117
|
+
|
|
118
|
+
- **Nothing the tool writes has been opened in Tableau yet.** That covers renamed workbooks and workbooks made from templates. The XML follows what Tableau writes, but open one on a copy and check it before relying on it. The original file is never modified or overwritten.
|
|
119
|
+
- **Only part of the R package is ported.** Missing: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, calculation complexity, the replication brief and the Shiny inspector. The GUI covers some of what the inspector did.
|
|
120
|
+
- Where the R version has a bug, this one does not copy it: joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
|
|
121
|
+
- The GUI is meant to run on your own machine for one person. It refuses requests that come from other websites, but there is no login, so do not put it on a shared network.
|
|
122
|
+
|
|
123
|
+
## Why a Tableau parser
|
|
124
|
+
|
|
125
|
+
A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python is not hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
|
|
126
|
+
|
|
127
|
+
## Contributing
|
|
128
|
+
|
|
129
|
+
Issues and pull requests are welcome at [github.com/DDSNA/py-tbparse](https://github.com/DDSNA/py-tbparse). The test setup is in the [development guide](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md). `AGENTS.md` describes the code layout and the rules for porting a function from the R package.
|
|
130
|
+
|
|
131
|
+
## Credit
|
|
132
|
+
|
|
133
|
+
Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
|
|
134
|
+
|
|
135
|
+
## License
|
|
136
|
+
|
|
137
|
+
MIT, see [LICENSE](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE).
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# py-tbparse
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
4
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
5
|
+
[](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml)
|
|
6
|
+
[](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE)
|
|
7
|
+
|
|
8
|
+
Read Tableau workbooks (`.twb` and `.twbx`) from Python. You get datasources, fields, calculated fields, joins, relationships and dashboards as pandas DataFrames. It is plain Python on top of `lxml` and `pandas`, so Tableau and R are not needed.
|
|
9
|
+
|
|
10
|
+
It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The command-line tool, the browser GUI, renaming, templates, workbook diffing and folder scanning are new here.
|
|
11
|
+
|
|
12
|
+

|
|
13
|
+
|
|
14
|
+
## What it does
|
|
15
|
+
|
|
16
|
+
- Lists what is in a workbook: datasources, parameters, fields, calculations, joins, relationships, dashboards, custom SQL, published sources.
|
|
17
|
+
- Checks relationships and finds calculations that refer to fields the workbook does not have.
|
|
18
|
+
- Shows which worksheets, dashboards and calculations use each field.
|
|
19
|
+
- Suggests clean names for fields, sheets, dashboards and other objects, and writes a renamed copy.
|
|
20
|
+
- Turns a finished workbook into a template you can fill with other data.
|
|
21
|
+
- Compares two workbooks, scans a folder of them, and draws the data model as a graph.
|
|
22
|
+
- Works from Python, from the command line, or in a local browser page.
|
|
23
|
+
|
|
24
|
+
## Install
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install py-tbparse
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Python 3.9 or newer. The package, the import and both commands all use the same name: `pip install py-tbparse`, `import py_tbparse`, `py-tbparse`, `py-tbparse-gui`. Releases up to 0.2.0 used `twbparser_py` and `twbparser`; those names are gone.
|
|
31
|
+
|
|
32
|
+
## Quick start
|
|
33
|
+
|
|
34
|
+
From Python:
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from py_tbparse import TwbParser
|
|
38
|
+
|
|
39
|
+
p = TwbParser("workbook.twb") # or a .twbx
|
|
40
|
+
p.get_overview() # counts of everything
|
|
41
|
+
p.get_calculated_fields() # each calculation and its formula
|
|
42
|
+
p.get_relationships() # how the tables connect
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Every getter returns a DataFrame. The others are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql`, `get_published_refs` and `get_field_usage`. `get_relationship_graph_dot()` returns the data model as Graphviz text and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
|
|
46
|
+
|
|
47
|
+
From the command line:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
py-tbparse workbook.twb # overview
|
|
51
|
+
py-tbparse workbook.twb calculated-fields
|
|
52
|
+
py-tbparse workbook.twb fields --format csv -o fields.csv
|
|
53
|
+
py-tbparse workbook.twb validate # exit code 2 if it finds a problem
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
In the browser:
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
py-tbparse-gui workbook.twb
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Across workbooks:
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from py_tbparse import TwbParser, diff_workbooks, scan_folder
|
|
66
|
+
|
|
67
|
+
diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
|
|
68
|
+
scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Guides
|
|
72
|
+
|
|
73
|
+
- [Command line](https://github.com/DDSNA/py-tbparse/blob/main/docs/cli.md): every table, the `diff`, `batch`, `rename` and `template` commands, and their options.
|
|
74
|
+
- [Browser GUI](https://github.com/DDSNA/py-tbparse/blob/main/docs/gui.md): opening files, the table, the overview, the graph, themes.
|
|
75
|
+
- [Renaming](https://github.com/DDSNA/py-tbparse/blob/main/docs/renaming.md): clean names after a datasource switch, editing the suggestions, what will stay broken.
|
|
76
|
+
- [Templates](https://github.com/DDSNA/py-tbparse/blob/main/docs/templates.md): make a template from a workbook and apply it to new data.
|
|
77
|
+
- [`.twbx` files](https://github.com/DDSNA/py-tbparse/blob/main/docs/twbx.md): they are read straight from the zip; how to extract the contents.
|
|
78
|
+
- [Development](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md): running the tests, the workbook corpus, the browser tests.
|
|
79
|
+
|
|
80
|
+
## Limits
|
|
81
|
+
|
|
82
|
+
- **Nothing the tool writes has been opened in Tableau yet.** That covers renamed workbooks and workbooks made from templates. The XML follows what Tableau writes, but open one on a copy and check it before relying on it. The original file is never modified or overwritten.
|
|
83
|
+
- **Only part of the R package is ported.** Missing: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, calculation complexity, the replication brief and the Shiny inspector. The GUI covers some of what the inspector did.
|
|
84
|
+
- Where the R version has a bug, this one does not copy it: joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
|
|
85
|
+
- The GUI is meant to run on your own machine for one person. It refuses requests that come from other websites, but there is no login, so do not put it on a shared network.
|
|
86
|
+
|
|
87
|
+
## Why a Tableau parser
|
|
88
|
+
|
|
89
|
+
A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python is not hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
|
|
90
|
+
|
|
91
|
+
## Contributing
|
|
92
|
+
|
|
93
|
+
Issues and pull requests are welcome at [github.com/DDSNA/py-tbparse](https://github.com/DDSNA/py-tbparse). The test setup is in the [development guide](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md). `AGENTS.md` describes the code layout and the rules for porting a function from the R package.
|
|
94
|
+
|
|
95
|
+
## Credit
|
|
96
|
+
|
|
97
|
+
Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
|
|
98
|
+
|
|
99
|
+
## License
|
|
100
|
+
|
|
101
|
+
MIT, see [LICENSE](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE).
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -10,7 +10,7 @@ from .dashboards import dashboard_sheets, list_dashboards
|
|
|
10
10
|
from .datasources import extract_datasource_details, extract_named_connections, extract_parameters
|
|
11
11
|
from .diff import diff_tables, diff_workbooks
|
|
12
12
|
from .fields import extract_columns_with_table_source, infer_implicit_relationships
|
|
13
|
-
from .graph import to_dot
|
|
13
|
+
from .graph import graph_data, to_dot
|
|
14
14
|
from .joins import extract_joins
|
|
15
15
|
from .parser import TwbParser
|
|
16
16
|
from .published import extract_published_refs
|
|
@@ -21,8 +21,20 @@ from .rename import (
|
|
|
21
21
|
load_rename_mapping,
|
|
22
22
|
normalize_name,
|
|
23
23
|
suggest_field_renames,
|
|
24
|
+
suggest_renames,
|
|
24
25
|
)
|
|
25
26
|
from .sql import extract_custom_sql, extract_initial_sql
|
|
27
|
+
from .templates import (
|
|
28
|
+
Template,
|
|
29
|
+
TemplateError,
|
|
30
|
+
apply_template,
|
|
31
|
+
load_mapping,
|
|
32
|
+
load_template,
|
|
33
|
+
make_template,
|
|
34
|
+
read_data,
|
|
35
|
+
suggest_mapping,
|
|
36
|
+
)
|
|
37
|
+
from .usage import field_usage
|
|
26
38
|
from .validators import validate_relationships
|
|
27
39
|
from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
|
|
28
40
|
|
|
@@ -39,6 +51,7 @@ __all__ = [
|
|
|
39
51
|
"infer_implicit_relationships",
|
|
40
52
|
"extract_joins",
|
|
41
53
|
"to_dot",
|
|
54
|
+
"graph_data",
|
|
42
55
|
"extract_relations",
|
|
43
56
|
"extract_relationships",
|
|
44
57
|
"extract_custom_sql",
|
|
@@ -56,6 +69,16 @@ __all__ = [
|
|
|
56
69
|
"load_rename_mapping",
|
|
57
70
|
"normalize_name",
|
|
58
71
|
"suggest_field_renames",
|
|
72
|
+
"suggest_renames",
|
|
73
|
+
"field_usage",
|
|
74
|
+
"Template",
|
|
75
|
+
"TemplateError",
|
|
76
|
+
"make_template",
|
|
77
|
+
"load_template",
|
|
78
|
+
"read_data",
|
|
79
|
+
"suggest_mapping",
|
|
80
|
+
"load_mapping",
|
|
81
|
+
"apply_template",
|
|
59
82
|
]
|
|
60
83
|
|
|
61
84
|
try:
|
|
@@ -39,12 +39,28 @@ def _calculated_fields(p: TwbParser, include_parameters: bool = False, **_kw) ->
|
|
|
39
39
|
|
|
40
40
|
|
|
41
41
|
def _field_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
|
|
42
|
-
datasource=None, **_kw) -> pd.DataFrame:
|
|
42
|
+
datasource=None, kinds=None, **_kw) -> pd.DataFrame:
|
|
43
|
+
if kinds: # the GUI's "everything in the report" switch
|
|
44
|
+
return p.get_renames(
|
|
45
|
+
reference=reference, style=style, only_changed=only_changed, datasource=datasource, kinds=kinds
|
|
46
|
+
)
|
|
43
47
|
return p.get_field_renames(
|
|
44
48
|
reference=reference, style=style, only_changed=only_changed, datasource=datasource
|
|
45
49
|
)
|
|
46
50
|
|
|
47
51
|
|
|
52
|
+
def _report_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
|
|
53
|
+
datasource=None, **_kw) -> pd.DataFrame:
|
|
54
|
+
return p.get_renames(reference=reference, style=style, only_changed=only_changed, datasource=datasource)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _field_usage(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
58
|
+
df = p.get_field_usage()
|
|
59
|
+
for col in ("sheets", "dashboards", "calculations"):
|
|
60
|
+
df[col] = df[col].map("; ".join)
|
|
61
|
+
return df
|
|
62
|
+
|
|
63
|
+
|
|
48
64
|
def _joins(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
49
65
|
return p.get_joins()
|
|
50
66
|
|
|
@@ -77,6 +93,10 @@ def _initial_sql(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
|
77
93
|
return p.get_initial_sql()
|
|
78
94
|
|
|
79
95
|
|
|
96
|
+
def _missing_references(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
97
|
+
return p.get_missing_references()
|
|
98
|
+
|
|
99
|
+
|
|
80
100
|
def _published_refs(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
81
101
|
return p.get_published_refs()
|
|
82
102
|
|
|
@@ -91,6 +111,9 @@ TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
|
|
|
91
111
|
"raw-fields": _raw_fields,
|
|
92
112
|
"calculated-fields": _calculated_fields,
|
|
93
113
|
"field-renames": _field_renames,
|
|
114
|
+
"report-renames": _report_renames,
|
|
115
|
+
"field-usage": _field_usage,
|
|
116
|
+
"missing-references": _missing_references,
|
|
94
117
|
"joins": _joins,
|
|
95
118
|
"relations": _relations,
|
|
96
119
|
"relationships": _relationships,
|