py-tbparse 0.4.0__tar.gz → 0.4.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/AGENTS.md +59 -12
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/MANIFEST.in +1 -0
- py_tbparse-0.4.4/PKG-INFO +137 -0
- py_tbparse-0.4.4/README.md +101 -0
- py_tbparse-0.4.4/docs/gui-field-renames.png +0 -0
- py_tbparse-0.4.4/docs/gui-graph.png +0 -0
- py_tbparse-0.4.4/docs/gui-overview.png +0 -0
- py_tbparse-0.4.4/docs/gui-themes.png +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/__init__.py +2 -1
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/_tables.py +5 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/graph.py +51 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/parser.py +22 -1
- py_tbparse-0.4.4/py_tbparse/report.py +124 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/usage.py +82 -0
- py_tbparse-0.4.4/py_tbparse/webgui.py +657 -0
- py_tbparse-0.4.4/py_tbparse/webui/app.css +318 -0
- py_tbparse-0.4.4/py_tbparse/webui/app.js +1497 -0
- py_tbparse-0.4.4/py_tbparse/webui/graph.js +453 -0
- py_tbparse-0.4.4/py_tbparse/webui/index.html +138 -0
- py_tbparse-0.4.4/py_tbparse/webui/table.js +359 -0
- py_tbparse-0.4.4/py_tbparse/webui/themes.css +237 -0
- py_tbparse-0.4.4/py_tbparse/webui/tokens.css +104 -0
- py_tbparse-0.4.4/py_tbparse.egg-info/PKG-INFO +137 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/SOURCES.txt +21 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/pyproject.toml +2 -2
- py_tbparse-0.4.4/tests/conftest.py +160 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_graph.py +27 -0
- py_tbparse-0.4.4/tests/test_gui_a11y.py +476 -0
- py_tbparse-0.4.4/tests/test_gui_browser.py +1068 -0
- py_tbparse-0.4.4/tests/test_gui_graph.py +401 -0
- py_tbparse-0.4.4/tests/test_gui_layout.py +304 -0
- py_tbparse-0.4.4/tests/test_gui_open.py +167 -0
- py_tbparse-0.4.4/tests/test_gui_overview.py +167 -0
- py_tbparse-0.4.4/tests/test_gui_table.py +1111 -0
- py_tbparse-0.4.4/tests/test_gui_themes.py +229 -0
- py_tbparse-0.4.4/tests/test_overview_endpoint.py +81 -0
- py_tbparse-0.4.4/tests/test_report.py +214 -0
- py_tbparse-0.4.4/tests/test_upload.py +189 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_webgui.py +83 -18
- py_tbparse-0.4.4/tests/test_webui_tokens.py +175 -0
- py_tbparse-0.4.0/PKG-INFO +0 -250
- py_tbparse-0.4.0/README.md +0 -214
- py_tbparse-0.4.0/docs/gui-field-renames.png +0 -0
- py_tbparse-0.4.0/docs/gui-overview.png +0 -0
- py_tbparse-0.4.0/py_tbparse/webgui.py +0 -1124
- py_tbparse-0.4.0/py_tbparse.egg-info/PKG-INFO +0 -250
- py_tbparse-0.4.0/tests/conftest.py +0 -56
- py_tbparse-0.4.0/tests/test_gui_browser.py +0 -475
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/LICENSE +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/_clean.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/_xml.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/batch.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/calculated_fields.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/cli.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/dashboards.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/datasources.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/diff.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/fields.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/joins.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/published.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/py.typed +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/relationships.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/rename.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/sql.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/templates.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse/validators.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/dependency_links.txt +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/entry_points.txt +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/requires.txt +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/top_level.txt +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/setup.cfg +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/fixtures/test_for_wenjie.twb +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/fixtures/test_for_zip.twbx +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_batch.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_calculated_fields.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_clean.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_cli.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_corpus.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_dashboards.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_datasources.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_diff.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_fields.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_joins.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_parser.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_public_workbooks.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_published.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_relationships.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_rename.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_rename_all.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_rename_mapping.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_sql.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_templates.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_validators.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_version.py +0 -0
- {py_tbparse-0.4.0 → py_tbparse-0.4.4}/tests/test_xml.py +0 -0
|
@@ -122,10 +122,11 @@ beyond the R package's scope:
|
|
|
122
122
|
|---|---|
|
|
123
123
|
| `_tables.py` | Name → `TwbParser`-accessor registry shared by `cli.py`, `webgui.py`, `diff.py`, and `batch.py`. Adding a new extractor to `TwbParser`? Add it here too so it's automatically available everywhere else. |
|
|
124
124
|
| `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename`/`template` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
|
|
125
|
-
| `webgui.py` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
|
|
126
|
-
| `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
|
|
125
|
+
| `webgui.py` + `webui/` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. The server is `webgui.py`; the page is `webui/` (`index.html`, `tokens.css`, `themes.css`, `app.css`, `table.js`, `graph.js`, `app.js`). Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
|
|
126
|
+
| `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges); `graph_data()` is the same edges as plain data, which the GUI lays out and draws itself (`webui/graph.js`: layered layout, SVG, pan/zoom, keyboard). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
|
|
127
127
|
| `diff.py` | `diff_tables()`/`diff_workbooks()`: row-level added/removed diff between two workbooks' same-named table, via `_tables.TABLE_SPECS`. No "changed" classification without a natural key — a changed row shows as one removed + one added row. |
|
|
128
128
|
| `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `suggest_renames()` (the same for every kind of object: field, parameter, worksheet, dashboard, datasource, folder, hierarchy; adds a `kind` column), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` and `report-renames` tables in `_tables.py`. A worksheet/dashboard rename must rewrite every reference (`_SHEET_REFERENCES`); if you learn of another place Tableau writes a sheet name, add it there. |
|
|
129
|
+
| `report.py` | `workbook_report()`: the overview's report card (summary sentence, health checks, dashboards with their sheets) built from `validate_relationships`, `field_usage` and `usage.missing_references`. Every health item names the table and column filters that list exactly the rows it counted; a test keeps count and rows equal on all 200 corpus workbooks. |
|
|
129
130
|
| `usage.py` | `field_usage()`: for every field, the worksheets, dashboards and calculations that use it, followed through calculations, groups/sets and bins (the `field-usage` table). Python-native rather than a port of the R package's field-usage helper. |
|
|
130
131
|
| `templates.py` | Workbook templates. `make_template()` writes a `.twbx` with a `template.json` manifest (required fields from `usage.py`, parameters, connections; data, extracts and credentials stripped); `read_data()` describes new data (CSV, or a `.twb`/`.twbx`/`.tds`); `suggest_mapping()` pairs fields with columns (reuses `rename._match_key`, type checks against the field's *physical* type, `datatype-customized` fields keep their type); `apply_template()` replaces the datasource's connection, keeping every field's local name, sets parameters and stores `template-answers.json`. A CSV connection is written in the 2020.2+ object-model shape (both `_.fcp.ObjectModelEncapsulateLegacy` relations, object graph, table column, `object-id` per record) when the template uses it -- keep those in step if you touch one. Untested in Tableau itself. |
|
|
131
132
|
| `batch.py` | `scan_folder()`: runs one table across every `.twb`/`.twbx` in a directory, concatenated with a `workbook` column. Skips (with a warning) any file that fails to load/extract rather than aborting the batch. |
|
|
@@ -203,21 +204,38 @@ string became a literal newline, splitting a JS string literal across
|
|
|
203
204
|
two lines) killed the entire script, so no handlers bound and the UI was
|
|
204
205
|
inert — with the whole suite green.
|
|
205
206
|
|
|
206
|
-
So: **any change to
|
|
207
|
+
So: **any change to the page (`py_tbparse/webui/`) needs a browser test**, in
|
|
208
|
+
The browser suite is split by concern, all sharing the fixtures in `tests/test_gui_browser.py`:
|
|
209
|
+
`test_gui_table.py` (windowing invariants, pipeline vs. a Python oracle, a seeded random walk;
|
|
210
|
+
`PYTBPARSE_WALK_SEEDS`/`PYTBPARSE_WALK_STEPS` widen it), `test_gui_a11y.py` (roles, keyboard, focus, rendered
|
|
211
|
+
contrast in both themes) and `test_gui_layout.py` (no sideways overflow from 320 px up, layout stability).
|
|
212
|
+
Design decisions the tests pin: only sorting uses a view transition (Chromium sends clicks to the page root while
|
|
213
|
+
one runs); column `MIN_WIDTH` is 80; per-table view settings are keyed by column name; disabled menu items use
|
|
214
|
+
`aria-disabled` so they stay focusable.
|
|
215
|
+
|
|
207
216
|
`tests/test_gui_browser.py`, which runs the page in real headless
|
|
208
217
|
Chromium via Playwright and fails on any uncaught JS error. The cheap
|
|
209
218
|
structural guards in `test_webgui.py` (unterminated string literals,
|
|
210
219
|
bracket balance) are a backstop, not a substitute.
|
|
211
220
|
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
(
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
the
|
|
221
|
+
The page is real files, not a Python string: `py_tbparse/webui/index.html`, `tokens.css` (the design
|
|
222
|
+
tokens), `app.css` and `app.js`. They are served by `webgui.py` from a fixed whitelist under
|
|
223
|
+
`/static/` (`_ASSETS`), so a request can never reach any other file; add a new asset to
|
|
224
|
+
`_ASSETS` and to the `webui/*` package-data (pyproject and MANIFEST.in) or it will not ship.
|
|
225
|
+
`index.html` is the only thing that gets server values: `_render_index()` replaces
|
|
226
|
+
`<!--APP_CONFIG-->` (now in `<head>`, so the saved theme is on `<html>` before the first paint) with the single
|
|
227
|
+
inline `<script>`. Keep it that way (one inline script plus the three external ones, `table.js`, `graph.js`, then
|
|
228
|
+
`app.js`); a test counts them. `populateTables();` must appear exactly once
|
|
229
|
+
in `app.js`, and the structural tests (unterminated string literals, bracket balance) read both served scripts.
|
|
230
|
+
Keep apostrophes out of JS strings and comments (the test counts quotes per line).
|
|
231
|
+
|
|
232
|
+
`POST /upload` (drag and drop, Open file) takes raw bytes with `Content-Type: application/octet-stream` and the name
|
|
233
|
+
in `X-Filename`: neither is CORS-safelisted, so another site cannot send it without a preflight. It streams to a
|
|
234
|
+
`mkdtemp` directory, refuses over `MAX_UPLOAD_BYTES` (413) and content that does not match the extension, and keeps
|
|
235
|
+
only one upload at a time. An uploaded workbook cannot "create beside the original" (409), only download.
|
|
236
|
+
|
|
237
|
+
Any server-side value spliced into the page (currently `TABLE_NAMES`, the preloaded workbook path
|
|
238
|
+
and the version, all in `_render_index()`'s config script) must go through `_json_for_script()`, not
|
|
221
239
|
bare `json.dumps()`. `json.dumps` doesn't escape `/`, so a value
|
|
222
240
|
containing the literal text `</script>` closes the script tag early in
|
|
223
241
|
the browser's HTML parser — this is real, not theoretical, since the
|
|
@@ -241,6 +259,35 @@ leaves inputs empty and tests time out) and may crash rendering. Don't add
|
|
|
241
259
|
startup, which makes collection fail for anyone who doesn't have it
|
|
242
260
|
installed.
|
|
243
261
|
|
|
262
|
+
The table is windowed (`table.js`, class `VTable`): rows have a fixed height (`--row-h`) and only the rows
|
|
263
|
+
near the viewport exist in the page, so code and tests that read rows from the DOM see a window, not the table.
|
|
264
|
+
Use `aria-rowcount` for the size and the column menu's "Copy column values" to read a whole column (the tests do:
|
|
265
|
+
`_copy_column`). Never render every row; the old table needed over 4 s to sort 20,000 rows and could not draw
|
|
266
|
+
50,000. Data work (search index, typed sort) is in `app.js`; each render records the User Timing measure
|
|
267
|
+
`py-tbparse:table`, which the speed tests read. The budgets (first paint of 1,000 rows 150 ms, filtering 50,000 rows
|
|
268
|
+
100 ms) are enforced at twice the figure. Heavy tests skip below 1.5 GB of free memory (`_enough_memory`); this
|
|
269
|
+
sandbox has only 2 CPUs and 7.9 GB, and a 50,000-row DOM render has crashed it before.
|
|
270
|
+
|
|
271
|
+
The page's colours, spacing, type and motion are tokens in `webui/tokens.css`; use them rather than literals.
|
|
272
|
+
Colour themes are `webui/themes.css`: each theme is a complete colour set in light and dark, selected by
|
|
273
|
+
`<html data-theme data-mode>` (Auto is resolved to light/dark by the head script and kept in step by `app.js`;
|
|
274
|
+
dark colours live under `[data-mode="dark"]`, there is no `prefers-color-scheme` query). To add a theme, add its
|
|
275
|
+
two blocks there, its id to `webgui.THEMES` and its label to `THEME_LABELS` in `app.js`; the token test checks it.
|
|
276
|
+
`tests/test_webui_tokens.py` reads those files and fails if any text/background pair drops below WCAG AA
|
|
277
|
+
4.5:1, if the stylesheet uses an undefined variable, or if text is faded with `opacity` (use `--faint`).
|
|
278
|
+
Motion durations come from the `--dur-*` tokens, which `prefers-reduced-motion` sets to instant; keep new
|
|
279
|
+
animations on those tokens.
|
|
280
|
+
|
|
281
|
+
`scripts/readme_screenshots.py` retakes the README's four images from the made-up `docs/demo/coffee-shop.twb`
|
|
282
|
+
(built by `scripts/make_demo_workbook.py`, which is the one place that workbook is defined); run both after a
|
|
283
|
+
visible change.
|
|
284
|
+
|
|
285
|
+
`scripts/gui_screenshots.py WORKBOOK OUT_DIR [--compare BASELINE_DIR]` captures nine GUI states (start,
|
|
286
|
+
overview and fields in light and dark, renames, graph, phone width) deterministically. Use it for GUI
|
|
287
|
+
refactors: a pure refactor must compare all-identical to the baseline taken before it; a redesign is expected to
|
|
288
|
+
differ, so review the new look and re-capture the baseline. The redesign plan and its decisions are in
|
|
289
|
+
`docs/ui-redesign-plan.md`.
|
|
290
|
+
|
|
244
291
|
## Commit / PR conventions
|
|
245
292
|
|
|
246
293
|
Nothing project-specific beyond the harness defaults — see repo commit
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: py-tbparse
|
|
3
|
+
Version: 0.4.4
|
|
4
|
+
Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
|
|
5
|
+
Author: DDSNA
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/DDSNA/py-tbparse
|
|
8
|
+
Project-URL: Source, https://github.com/DDSNA/py-tbparse
|
|
9
|
+
Project-URL: Issues, https://github.com/DDSNA/py-tbparse/issues
|
|
10
|
+
Keywords: tableau,twb,twbx,workbook,parser,pandas
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
+
Classifier: Topic :: Office/Business
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Requires-Dist: lxml>=4.9
|
|
27
|
+
Requires-Dist: pandas>=1.5
|
|
28
|
+
Provides-Extra: test
|
|
29
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
30
|
+
Provides-Extra: browser
|
|
31
|
+
Requires-Dist: playwright>=1.40; extra == "browser"
|
|
32
|
+
Provides-Extra: dev
|
|
33
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
34
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# py-tbparse
|
|
38
|
+
|
|
39
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
40
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
41
|
+
[](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml)
|
|
42
|
+
[](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE)
|
|
43
|
+
|
|
44
|
+
Read Tableau workbooks (`.twb` and `.twbx`) from Python. You get datasources, fields, calculated fields, joins, relationships and dashboards as pandas DataFrames. It is plain Python on top of `lxml` and `pandas`, so Tableau and R are not needed.
|
|
45
|
+
|
|
46
|
+
It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The command-line tool, the browser GUI, renaming, templates, workbook diffing and folder scanning are new here.
|
|
47
|
+
|
|
48
|
+

|
|
49
|
+
|
|
50
|
+
## What it does
|
|
51
|
+
|
|
52
|
+
- Lists what is in a workbook: datasources, parameters, fields, calculations, joins, relationships, dashboards, custom SQL, published sources.
|
|
53
|
+
- Checks relationships and finds calculations that refer to fields the workbook does not have.
|
|
54
|
+
- Shows which worksheets, dashboards and calculations use each field.
|
|
55
|
+
- Suggests clean names for fields, sheets, dashboards and other objects, and writes a renamed copy.
|
|
56
|
+
- Turns a finished workbook into a template you can fill with other data.
|
|
57
|
+
- Compares two workbooks, scans a folder of them, and draws the data model as a graph.
|
|
58
|
+
- Works from Python, from the command line, or in a local browser page.
|
|
59
|
+
|
|
60
|
+
## Install
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
pip install py-tbparse
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Python 3.9 or newer. The package, the import and both commands all use the same name: `pip install py-tbparse`, `import py_tbparse`, `py-tbparse`, `py-tbparse-gui`. Releases up to 0.2.0 used `twbparser_py` and `twbparser`; those names are gone.
|
|
67
|
+
|
|
68
|
+
## Quick start
|
|
69
|
+
|
|
70
|
+
From Python:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from py_tbparse import TwbParser
|
|
74
|
+
|
|
75
|
+
p = TwbParser("workbook.twb") # or a .twbx
|
|
76
|
+
p.get_overview() # counts of everything
|
|
77
|
+
p.get_calculated_fields() # each calculation and its formula
|
|
78
|
+
p.get_relationships() # how the tables connect
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Every getter returns a DataFrame. The others are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql`, `get_published_refs` and `get_field_usage`. `get_relationship_graph_dot()` returns the data model as Graphviz text and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
|
|
82
|
+
|
|
83
|
+
From the command line:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
py-tbparse workbook.twb # overview
|
|
87
|
+
py-tbparse workbook.twb calculated-fields
|
|
88
|
+
py-tbparse workbook.twb fields --format csv -o fields.csv
|
|
89
|
+
py-tbparse workbook.twb validate # exit code 2 if it finds a problem
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
In the browser:
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
py-tbparse-gui workbook.twb
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Across workbooks:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
from py_tbparse import TwbParser, diff_workbooks, scan_folder
|
|
102
|
+
|
|
103
|
+
diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
|
|
104
|
+
scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## Guides
|
|
108
|
+
|
|
109
|
+
- [Command line](https://github.com/DDSNA/py-tbparse/blob/main/docs/cli.md): every table, the `diff`, `batch`, `rename` and `template` commands, and their options.
|
|
110
|
+
- [Browser GUI](https://github.com/DDSNA/py-tbparse/blob/main/docs/gui.md): opening files, the table, the overview, the graph, themes.
|
|
111
|
+
- [Renaming](https://github.com/DDSNA/py-tbparse/blob/main/docs/renaming.md): clean names after a datasource switch, editing the suggestions, what will stay broken.
|
|
112
|
+
- [Templates](https://github.com/DDSNA/py-tbparse/blob/main/docs/templates.md): make a template from a workbook and apply it to new data.
|
|
113
|
+
- [`.twbx` files](https://github.com/DDSNA/py-tbparse/blob/main/docs/twbx.md): they are read straight from the zip; how to extract the contents.
|
|
114
|
+
- [Development](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md): running the tests, the workbook corpus, the browser tests.
|
|
115
|
+
|
|
116
|
+
## Limits
|
|
117
|
+
|
|
118
|
+
- **Nothing the tool writes has been opened in Tableau yet.** That covers renamed workbooks and workbooks made from templates. The XML follows what Tableau writes, but open one on a copy and check it before relying on it. The original file is never modified or overwritten.
|
|
119
|
+
- **Only part of the R package is ported.** Missing: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, calculation complexity, the replication brief and the Shiny inspector. The GUI covers some of what the inspector did.
|
|
120
|
+
- Where the R version has a bug, this one does not copy it: joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
|
|
121
|
+
- The GUI is meant to run on your own machine for one person. It refuses requests that come from other websites, but there is no login, so do not put it on a shared network.
|
|
122
|
+
|
|
123
|
+
## Why a Tableau parser
|
|
124
|
+
|
|
125
|
+
A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python is not hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
|
|
126
|
+
|
|
127
|
+
## Contributing
|
|
128
|
+
|
|
129
|
+
Issues and pull requests are welcome at [github.com/DDSNA/py-tbparse](https://github.com/DDSNA/py-tbparse). The test setup is in the [development guide](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md). `AGENTS.md` describes the code layout and the rules for porting a function from the R package.
|
|
130
|
+
|
|
131
|
+
## Credit
|
|
132
|
+
|
|
133
|
+
Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
|
|
134
|
+
|
|
135
|
+
## License
|
|
136
|
+
|
|
137
|
+
MIT, see [LICENSE](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE).
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# py-tbparse
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
4
|
+
[](https://pypi.org/project/py-tbparse/)
|
|
5
|
+
[](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml)
|
|
6
|
+
[](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE)
|
|
7
|
+
|
|
8
|
+
Read Tableau workbooks (`.twb` and `.twbx`) from Python. You get datasources, fields, calculated fields, joins, relationships and dashboards as pandas DataFrames. It is plain Python on top of `lxml` and `pandas`, so Tableau and R are not needed.
|
|
9
|
+
|
|
10
|
+
It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The command-line tool, the browser GUI, renaming, templates, workbook diffing and folder scanning are new here.
|
|
11
|
+
|
|
12
|
+

|
|
13
|
+
|
|
14
|
+
## What it does
|
|
15
|
+
|
|
16
|
+
- Lists what is in a workbook: datasources, parameters, fields, calculations, joins, relationships, dashboards, custom SQL, published sources.
|
|
17
|
+
- Checks relationships and finds calculations that refer to fields the workbook does not have.
|
|
18
|
+
- Shows which worksheets, dashboards and calculations use each field.
|
|
19
|
+
- Suggests clean names for fields, sheets, dashboards and other objects, and writes a renamed copy.
|
|
20
|
+
- Turns a finished workbook into a template you can fill with other data.
|
|
21
|
+
- Compares two workbooks, scans a folder of them, and draws the data model as a graph.
|
|
22
|
+
- Works from Python, from the command line, or in a local browser page.
|
|
23
|
+
|
|
24
|
+
## Install
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install py-tbparse
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Python 3.9 or newer. The package, the import and both commands all use the same name: `pip install py-tbparse`, `import py_tbparse`, `py-tbparse`, `py-tbparse-gui`. Releases up to 0.2.0 used `twbparser_py` and `twbparser`; those names are gone.
|
|
31
|
+
|
|
32
|
+
## Quick start
|
|
33
|
+
|
|
34
|
+
From Python:
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from py_tbparse import TwbParser
|
|
38
|
+
|
|
39
|
+
p = TwbParser("workbook.twb") # or a .twbx
|
|
40
|
+
p.get_overview() # counts of everything
|
|
41
|
+
p.get_calculated_fields() # each calculation and its formula
|
|
42
|
+
p.get_relationships() # how the tables connect
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Every getter returns a DataFrame. The others are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql`, `get_published_refs` and `get_field_usage`. `get_relationship_graph_dot()` returns the data model as Graphviz text and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
|
|
46
|
+
|
|
47
|
+
From the command line:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
py-tbparse workbook.twb # overview
|
|
51
|
+
py-tbparse workbook.twb calculated-fields
|
|
52
|
+
py-tbparse workbook.twb fields --format csv -o fields.csv
|
|
53
|
+
py-tbparse workbook.twb validate # exit code 2 if it finds a problem
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
In the browser:
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
py-tbparse-gui workbook.twb
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Across workbooks:
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from py_tbparse import TwbParser, diff_workbooks, scan_folder
|
|
66
|
+
|
|
67
|
+
diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
|
|
68
|
+
scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Guides
|
|
72
|
+
|
|
73
|
+
- [Command line](https://github.com/DDSNA/py-tbparse/blob/main/docs/cli.md): every table, the `diff`, `batch`, `rename` and `template` commands, and their options.
|
|
74
|
+
- [Browser GUI](https://github.com/DDSNA/py-tbparse/blob/main/docs/gui.md): opening files, the table, the overview, the graph, themes.
|
|
75
|
+
- [Renaming](https://github.com/DDSNA/py-tbparse/blob/main/docs/renaming.md): clean names after a datasource switch, editing the suggestions, what will stay broken.
|
|
76
|
+
- [Templates](https://github.com/DDSNA/py-tbparse/blob/main/docs/templates.md): make a template from a workbook and apply it to new data.
|
|
77
|
+
- [`.twbx` files](https://github.com/DDSNA/py-tbparse/blob/main/docs/twbx.md): they are read straight from the zip; how to extract the contents.
|
|
78
|
+
- [Development](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md): running the tests, the workbook corpus, the browser tests.
|
|
79
|
+
|
|
80
|
+
## Limits
|
|
81
|
+
|
|
82
|
+
- **Nothing the tool writes has been opened in Tableau yet.** That covers renamed workbooks and workbooks made from templates. The XML follows what Tableau writes, but open one on a copy and check it before relying on it. The original file is never modified or overwritten.
|
|
83
|
+
- **Only part of the R package is ported.** Missing: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, calculation complexity, the replication brief and the Shiny inspector. The GUI covers some of what the inspector did.
|
|
84
|
+
- Where the R version has a bug, this one does not copy it: joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
|
|
85
|
+
- The GUI is meant to run on your own machine for one person. It refuses requests that come from other websites, but there is no login, so do not put it on a shared network.
|
|
86
|
+
|
|
87
|
+
## Why a Tableau parser
|
|
88
|
+
|
|
89
|
+
A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python is not hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
|
|
90
|
+
|
|
91
|
+
## Contributing
|
|
92
|
+
|
|
93
|
+
Issues and pull requests are welcome at [github.com/DDSNA/py-tbparse](https://github.com/DDSNA/py-tbparse). The test setup is in the [development guide](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md). `AGENTS.md` describes the code layout and the rules for porting a function from the R package.
|
|
94
|
+
|
|
95
|
+
## Credit
|
|
96
|
+
|
|
97
|
+
Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
|
|
98
|
+
|
|
99
|
+
## License
|
|
100
|
+
|
|
101
|
+
MIT, see [LICENSE](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE).
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -10,7 +10,7 @@ from .dashboards import dashboard_sheets, list_dashboards
|
|
|
10
10
|
from .datasources import extract_datasource_details, extract_named_connections, extract_parameters
|
|
11
11
|
from .diff import diff_tables, diff_workbooks
|
|
12
12
|
from .fields import extract_columns_with_table_source, infer_implicit_relationships
|
|
13
|
-
from .graph import to_dot
|
|
13
|
+
from .graph import graph_data, to_dot
|
|
14
14
|
from .joins import extract_joins
|
|
15
15
|
from .parser import TwbParser
|
|
16
16
|
from .published import extract_published_refs
|
|
@@ -51,6 +51,7 @@ __all__ = [
|
|
|
51
51
|
"infer_implicit_relationships",
|
|
52
52
|
"extract_joins",
|
|
53
53
|
"to_dot",
|
|
54
|
+
"graph_data",
|
|
54
55
|
"extract_relations",
|
|
55
56
|
"extract_relationships",
|
|
56
57
|
"extract_custom_sql",
|
|
@@ -93,6 +93,10 @@ def _initial_sql(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
|
93
93
|
return p.get_initial_sql()
|
|
94
94
|
|
|
95
95
|
|
|
96
|
+
def _missing_references(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
97
|
+
return p.get_missing_references()
|
|
98
|
+
|
|
99
|
+
|
|
96
100
|
def _published_refs(p: TwbParser, **_kw) -> pd.DataFrame:
|
|
97
101
|
return p.get_published_refs()
|
|
98
102
|
|
|
@@ -109,6 +113,7 @@ TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
|
|
|
109
113
|
"field-renames": _field_renames,
|
|
110
114
|
"report-renames": _report_renames,
|
|
111
115
|
"field-usage": _field_usage,
|
|
116
|
+
"missing-references": _missing_references,
|
|
112
117
|
"joins": _joins,
|
|
113
118
|
"relations": _relations,
|
|
114
119
|
"relationships": _relationships,
|
|
@@ -44,6 +44,57 @@ def _inferred_label(row) -> str:
|
|
|
44
44
|
return f"{row.left_field} ~ {row.right_field} ({row.reason})"
|
|
45
45
|
|
|
46
46
|
|
|
47
|
+
def _collect(joins_df, relationships_df, inferred_df) -> list[tuple[str, str, str, str]]:
|
|
48
|
+
"""Every edge as (left table, right table, label, kind) with kind `join`, `relationship` or `inferred`,
|
|
49
|
+
in the order they are drawn: joins, relationships, then inferred guesses."""
|
|
50
|
+
edges = []
|
|
51
|
+
for left, right, label in _edges_from(joins_df, _join_label):
|
|
52
|
+
edges.append((left, right, label, "join"))
|
|
53
|
+
for left, right, label in _edges_from(relationships_df, _relationship_label):
|
|
54
|
+
edges.append((left, right, label, "relationship"))
|
|
55
|
+
if inferred_df is not None:
|
|
56
|
+
for left, right, label in _edges_from(inferred_df, _inferred_label):
|
|
57
|
+
edges.append((left, right, label, "inferred"))
|
|
58
|
+
return edges
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def graph_data(
|
|
62
|
+
joins_df: pd.DataFrame,
|
|
63
|
+
relationships_df: pd.DataFrame,
|
|
64
|
+
inferred_df: pd.DataFrame | None = None,
|
|
65
|
+
) -> dict:
|
|
66
|
+
"""The same graph as `to_dot`, as plain data for the GUI to lay out and draw:
|
|
67
|
+
`{"nodes": [{"id": ...}], "edges": [{"source", "target", "label", "kind"}]}`.
|
|
68
|
+
Nodes are sorted by name; edges keep `to_dot`'s order."""
|
|
69
|
+
edges = _collect(joins_df, relationships_df, inferred_df)
|
|
70
|
+
nodes = sorted({n for left, right, _, _ in edges for n in (left, right)})
|
|
71
|
+
return {
|
|
72
|
+
"nodes": [{"id": n} for n in nodes],
|
|
73
|
+
"edges": [{"source": l, "target": r, "label": lab, "kind": k} for l, r, lab, k in edges],
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def to_dot(
|
|
78
|
+
joins_df: pd.DataFrame,
|
|
79
|
+
relationships_df: pd.DataFrame,
|
|
80
|
+
inferred_df: pd.DataFrame | None = None,
|
|
81
|
+
graph_name: str = "twb",
|
|
82
|
+
) -> str:
|
|
83
|
+
"""Render join/relationship (and optionally inferred) edges as a
|
|
84
|
+
Graphviz DOT digraph string.
|
|
85
|
+
|
|
86
|
+
Solid edges are real joins/relationships; dashed edges (when
|
|
87
|
+
`inferred_df` is passed) are `infer_implicit_relationships()` guesses.
|
|
88
|
+
"""
|
|
89
|
+
edges = [(l, r, lab, kind == "inferred") for l, r, lab, kind in _collect(joins_df, relationships_df, inferred_df)]
|
|
90
|
+
|
|
91
|
+
nodes = sorted({n for left, right, _, _ in edges for n in (left, right)})
|
|
92
|
+
return {
|
|
93
|
+
"nodes": [{"id": n} for n in nodes],
|
|
94
|
+
"edges": [{"source": l, "target": r, "label": lab, "kind": k} for l, r, lab, k in edges],
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
47
98
|
def to_dot(
|
|
48
99
|
joins_df: pd.DataFrame,
|
|
49
100
|
relationships_df: pd.DataFrame,
|
|
@@ -20,7 +20,7 @@ from .datasources import (
|
|
|
20
20
|
extract_datasource_details,
|
|
21
21
|
)
|
|
22
22
|
from .fields import _FIELDS_COLUMNS, _INFERRED_COLUMNS, extract_columns_with_table_source, infer_implicit_relationships
|
|
23
|
-
from .graph import to_dot
|
|
23
|
+
from .graph import graph_data, to_dot
|
|
24
24
|
from .joins import _JOIN_COLUMNS, extract_joins
|
|
25
25
|
from .published import _PUBLISHED_COLUMNS, extract_published_refs
|
|
26
26
|
from .relationships import _RELATION_COLUMNS, _RELATIONSHIP_COLUMNS, extract_relations, extract_relationships
|
|
@@ -202,6 +202,14 @@ class TwbParser:
|
|
|
202
202
|
self.get_inferred_relationships() if include_inferred else None,
|
|
203
203
|
)
|
|
204
204
|
|
|
205
|
+
def get_relationship_graph_data(self, include_inferred: bool = False) -> dict:
|
|
206
|
+
"""The relationship graph as `{"nodes": [...], "edges": [...]}` (see `graph.graph_data`)."""
|
|
207
|
+
return graph_data(
|
|
208
|
+
self.get_joins(),
|
|
209
|
+
self.get_relationships(),
|
|
210
|
+
self.get_inferred_relationships() if include_inferred else None,
|
|
211
|
+
)
|
|
212
|
+
|
|
205
213
|
def get_field_renames(self, reference=None, **kwargs) -> pd.DataFrame:
|
|
206
214
|
"""Suggested clean field names; see `rename.suggest_field_renames`."""
|
|
207
215
|
from .rename import suggest_field_renames
|
|
@@ -215,6 +223,19 @@ class TwbParser:
|
|
|
215
223
|
|
|
216
224
|
return field_usage(self)
|
|
217
225
|
|
|
226
|
+
def get_missing_references(self) -> pd.DataFrame:
|
|
227
|
+
"""Calculations that name a field the workbook does not have; see
|
|
228
|
+
`usage.missing_references`."""
|
|
229
|
+
from .usage import missing_references
|
|
230
|
+
|
|
231
|
+
return missing_references(self)
|
|
232
|
+
|
|
233
|
+
def get_report(self) -> dict:
|
|
234
|
+
"""The workbook report card (summary, health checks); see `report.workbook_report`."""
|
|
235
|
+
from .report import workbook_report
|
|
236
|
+
|
|
237
|
+
return workbook_report(self)
|
|
238
|
+
|
|
218
239
|
def get_renames(self, reference=None, **kwargs) -> pd.DataFrame:
|
|
219
240
|
"""Suggested clean names for everything in the report (fields,
|
|
220
241
|
parameters, worksheets, dashboards, datasources, folders,
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""A plain-language report card for a workbook: what is in it and what deserves a look.
|
|
2
|
+
|
|
3
|
+
Python-native (the R package has no equivalent). It only combines checks that already exist:
|
|
4
|
+
`validate_relationships`, `field_usage` and `missing_references`, plus counts of the things that are
|
|
5
|
+
information rather than problems. The GUI's overview shows it; each item carries the table (and column
|
|
6
|
+
filters) that lists exactly the rows it counted, so a count on the card is never a different number from
|
|
7
|
+
the rows it opens.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import os
|
|
13
|
+
|
|
14
|
+
from .parser import TwbParser
|
|
15
|
+
from .usage import missing_references
|
|
16
|
+
|
|
17
|
+
PROBLEM, WARNING, INFO = "problem", "warning", "info"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _plural(n: int, one: str, many: str | None = None) -> str:
|
|
21
|
+
return f"{n} {one if n == 1 else (many or one + 's')}"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _item(severity: str, key: str, title: str, detail: str, count: int, table: str, filters=None) -> dict:
|
|
25
|
+
return {"id": key, "severity": severity, "title": title, "detail": detail, "count": count,
|
|
26
|
+
"table": table, "filters": filters or []}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _sheets_by_dashboard(parser: TwbParser) -> list[dict]:
|
|
30
|
+
doc = parser.xml_doc
|
|
31
|
+
out = []
|
|
32
|
+
for db in doc.xpath("/workbook/dashboards/dashboard[@name]"):
|
|
33
|
+
seen: list[str] = []
|
|
34
|
+
for z in db.xpath(".//zone[@worksheet]"):
|
|
35
|
+
if z.get("worksheet") not in seen:
|
|
36
|
+
seen.append(z.get("worksheet"))
|
|
37
|
+
out.append({"name": db.get("name"), "sheets": seen})
|
|
38
|
+
return sorted(out, key=lambda d: d["name"].lower())
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def workbook_report(parser: TwbParser) -> dict:
|
|
42
|
+
"""`{summary, counts, worksheets, dashboards, health}` for a loaded workbook.
|
|
43
|
+
|
|
44
|
+
`health` is a list of items (`severity`, `title`, `detail`, `count`, `table`, `filters`), problems
|
|
45
|
+
first. An empty list means nothing was found."""
|
|
46
|
+
doc = parser.xml_doc
|
|
47
|
+
overview = parser.get_overview().iloc[0].to_dict()
|
|
48
|
+
overview.pop("file", None)
|
|
49
|
+
# a .twbx is named by the package the person opened, not by the .twb inside it
|
|
50
|
+
name = os.path.basename(getattr(parser, "twbx_path", None) or parser.path or "") or "This workbook"
|
|
51
|
+
sheets = sorted({w.get("name") for w in doc.xpath("/workbook/worksheets/worksheet[@name]")}, key=str.lower)
|
|
52
|
+
dashboards = _sheets_by_dashboard(parser)
|
|
53
|
+
on_dashboard = {s for d in dashboards for s in d["sheets"]}
|
|
54
|
+
|
|
55
|
+
parts = [_plural(len(dashboards), "dashboard"), _plural(len(sheets), "worksheet"),
|
|
56
|
+
_plural(int(overview["datasources"]), "datasource")]
|
|
57
|
+
summary = f"{name} has {parts[0]}, {parts[1]} and {parts[2]}."
|
|
58
|
+
|
|
59
|
+
health: list[dict] = []
|
|
60
|
+
|
|
61
|
+
from .validators import validate_relationships
|
|
62
|
+
issues = validate_relationships(parser)["issues"]
|
|
63
|
+
if "unknown_tables" in issues:
|
|
64
|
+
n = len(issues["unknown_tables"])
|
|
65
|
+
health.append(_item(PROBLEM, "unknown-tables", _plural(n, "relationship") + " point at a table the workbook does not have",
|
|
66
|
+
"Check the left and right table names against the datasources.", n, "relationships"))
|
|
67
|
+
if "unknown_fields" in issues:
|
|
68
|
+
n = len(issues["unknown_fields"])
|
|
69
|
+
health.append(_item(PROBLEM, "unknown-fields", _plural(n, "relationship") + " use a field that is not in the workbook",
|
|
70
|
+
"The join keys may have been renamed or removed.", n, "relationships"))
|
|
71
|
+
|
|
72
|
+
missing = missing_references(parser)
|
|
73
|
+
if len(missing):
|
|
74
|
+
n = len(missing)
|
|
75
|
+
health.append(_item(WARNING, "missing-references", _plural(n, "reference") + " in calculations to fields that do not exist",
|
|
76
|
+
"A calculation names a field the workbook does not have, so it will show an error in Tableau.",
|
|
77
|
+
n, "missing-references"))
|
|
78
|
+
|
|
79
|
+
usage = parser.get_field_usage()
|
|
80
|
+
unused = usage[~usage["used"].astype(bool)] if len(usage) else usage
|
|
81
|
+
calcs = unused[unused["kind"] == "calculated"] if len(unused) else unused
|
|
82
|
+
raw = unused[unused["kind"] == "physical"] if len(unused) else unused
|
|
83
|
+
if len(calcs):
|
|
84
|
+
n = len(calcs)
|
|
85
|
+
health.append(_item(WARNING, "unused-calculations", _plural(n, "calculation") + " no worksheet uses",
|
|
86
|
+
"Candidates to delete, unless another workbook or a published source needs them.",
|
|
87
|
+
n, "field-usage", [{"col": "kind", "text": "calculated"}, {"col": "used", "text": "false"}]))
|
|
88
|
+
if len(raw):
|
|
89
|
+
n = len(raw)
|
|
90
|
+
health.append(_item(INFO, "unused-fields", _plural(n, "field") + " no worksheet uses",
|
|
91
|
+
"Ordinary for a wide data source; useful when you are slimming one down.",
|
|
92
|
+
n, "field-usage", [{"col": "kind", "text": "physical"}, {"col": "used", "text": "false"}]))
|
|
93
|
+
|
|
94
|
+
loose = [s for s in sheets if s not in on_dashboard]
|
|
95
|
+
if loose and dashboards:
|
|
96
|
+
health.append(_item(INFO, "sheets-off-dashboards", _plural(len(loose), "worksheet") + " not on any dashboard",
|
|
97
|
+
", ".join(loose[:6]) + (" and more" if len(loose) > 6 else ""), len(loose), "")) # no table lists these: sheets with no zone
|
|
98
|
+
|
|
99
|
+
inferred = int(overview["inferred_relationships"])
|
|
100
|
+
if inferred:
|
|
101
|
+
health.append(_item(INFO, "inferred", _plural(inferred, "relationship") + " could be inferred from matching field names",
|
|
102
|
+
"They are not modelled in the workbook; this is only a hint.", inferred, "inferred-relationships"))
|
|
103
|
+
for key, getter, table, title in (
|
|
104
|
+
("custom-sql", parser.get_custom_sql, "custom-sql", "custom SQL"),
|
|
105
|
+
("initial-sql", parser.get_initial_sql, "initial-sql", "initial SQL"),
|
|
106
|
+
):
|
|
107
|
+
n = len(getter())
|
|
108
|
+
if n:
|
|
109
|
+
health.append(_item(INFO, key, f"{n} {title} {'entry' if n == 1 else 'entries'}", "", n, table))
|
|
110
|
+
|
|
111
|
+
# A datasource with no connection of its own is a published one, or the Parameters source that every
|
|
112
|
+
# workbook with parameters has. Only mention it when something other than Parameters is in the list; the
|
|
113
|
+
# count is still every row the link opens, Parameters included, and the detail says so.
|
|
114
|
+
pub = parser.get_published_refs()
|
|
115
|
+
likely = pub[pub["likely_published"].astype(bool)] if len(pub) else pub
|
|
116
|
+
if len(likely) and (likely["name"] != "Parameters").any():
|
|
117
|
+
n = len(likely)
|
|
118
|
+
health.append(_item(INFO, "published", _plural(n, "datasource") + " without a connection of its own",
|
|
119
|
+
"Published data sources look like this. So does the Parameters source, which is counted too.",
|
|
120
|
+
n, "published-refs", [{"col": "likely_published", "text": "true"}]))
|
|
121
|
+
|
|
122
|
+
order = {PROBLEM: 0, WARNING: 1, INFO: 2}
|
|
123
|
+
health.sort(key=lambda h: order[h["severity"]])
|
|
124
|
+
return {"summary": summary, "counts": overview, "worksheets": sheets, "dashboards": dashboards, "health": health}
|