py-tbparse 0.3.0__tar.gz → 0.4.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/AGENTS.md +72 -16
  2. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/MANIFEST.in +1 -0
  3. py_tbparse-0.4.4/PKG-INFO +137 -0
  4. py_tbparse-0.4.4/README.md +101 -0
  5. py_tbparse-0.4.4/docs/gui-field-renames.png +0 -0
  6. py_tbparse-0.4.4/docs/gui-graph.png +0 -0
  7. py_tbparse-0.4.4/docs/gui-overview.png +0 -0
  8. py_tbparse-0.4.4/docs/gui-themes.png +0 -0
  9. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/__init__.py +24 -1
  10. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/_tables.py +24 -1
  11. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/cli.py +124 -1
  12. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/graph.py +51 -0
  13. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/parser.py +37 -1
  14. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/rename.py +272 -33
  15. py_tbparse-0.4.4/py_tbparse/report.py +124 -0
  16. py_tbparse-0.4.4/py_tbparse/templates.py +855 -0
  17. py_tbparse-0.4.4/py_tbparse/usage.py +248 -0
  18. py_tbparse-0.4.4/py_tbparse/webgui.py +657 -0
  19. py_tbparse-0.4.4/py_tbparse/webui/app.css +318 -0
  20. py_tbparse-0.4.4/py_tbparse/webui/app.js +1497 -0
  21. py_tbparse-0.4.4/py_tbparse/webui/graph.js +453 -0
  22. py_tbparse-0.4.4/py_tbparse/webui/index.html +138 -0
  23. py_tbparse-0.4.4/py_tbparse/webui/table.js +359 -0
  24. py_tbparse-0.4.4/py_tbparse/webui/themes.css +237 -0
  25. py_tbparse-0.4.4/py_tbparse/webui/tokens.css +104 -0
  26. py_tbparse-0.4.4/py_tbparse.egg-info/PKG-INFO +137 -0
  27. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/SOURCES.txt +26 -0
  28. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/pyproject.toml +2 -2
  29. py_tbparse-0.4.4/tests/conftest.py +160 -0
  30. py_tbparse-0.4.4/tests/test_corpus.py +84 -0
  31. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_graph.py +27 -0
  32. py_tbparse-0.4.4/tests/test_gui_a11y.py +476 -0
  33. py_tbparse-0.4.4/tests/test_gui_browser.py +1068 -0
  34. py_tbparse-0.4.4/tests/test_gui_graph.py +401 -0
  35. py_tbparse-0.4.4/tests/test_gui_layout.py +304 -0
  36. py_tbparse-0.4.4/tests/test_gui_open.py +167 -0
  37. py_tbparse-0.4.4/tests/test_gui_overview.py +167 -0
  38. py_tbparse-0.4.4/tests/test_gui_table.py +1111 -0
  39. py_tbparse-0.4.4/tests/test_gui_themes.py +229 -0
  40. py_tbparse-0.4.4/tests/test_overview_endpoint.py +81 -0
  41. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_public_workbooks.py +18 -0
  42. py_tbparse-0.4.4/tests/test_rename_all.py +215 -0
  43. py_tbparse-0.4.4/tests/test_report.py +214 -0
  44. py_tbparse-0.4.4/tests/test_templates.py +415 -0
  45. py_tbparse-0.4.4/tests/test_upload.py +189 -0
  46. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_webgui.py +108 -18
  47. py_tbparse-0.4.4/tests/test_webui_tokens.py +175 -0
  48. py_tbparse-0.3.0/PKG-INFO +0 -196
  49. py_tbparse-0.3.0/README.md +0 -160
  50. py_tbparse-0.3.0/docs/gui-field-renames.png +0 -0
  51. py_tbparse-0.3.0/docs/gui-overview.png +0 -0
  52. py_tbparse-0.3.0/py_tbparse/webgui.py +0 -1111
  53. py_tbparse-0.3.0/py_tbparse.egg-info/PKG-INFO +0 -196
  54. py_tbparse-0.3.0/tests/conftest.py +0 -56
  55. py_tbparse-0.3.0/tests/test_gui_browser.py +0 -452
  56. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/LICENSE +0 -0
  57. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/_clean.py +0 -0
  58. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/_xml.py +0 -0
  59. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/batch.py +0 -0
  60. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/calculated_fields.py +0 -0
  61. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/dashboards.py +0 -0
  62. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/datasources.py +0 -0
  63. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/diff.py +0 -0
  64. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/fields.py +0 -0
  65. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/joins.py +0 -0
  66. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/published.py +0 -0
  67. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/py.typed +0 -0
  68. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/relationships.py +0 -0
  69. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/sql.py +0 -0
  70. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse/validators.py +0 -0
  71. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/dependency_links.txt +0 -0
  72. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/entry_points.txt +0 -0
  73. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/requires.txt +0 -0
  74. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/py_tbparse.egg-info/top_level.txt +0 -0
  75. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/setup.cfg +0 -0
  76. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/fixtures/test_for_wenjie.twb +0 -0
  77. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/fixtures/test_for_zip.twbx +0 -0
  78. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_batch.py +0 -0
  79. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_calculated_fields.py +0 -0
  80. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_clean.py +0 -0
  81. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_cli.py +0 -0
  82. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_dashboards.py +0 -0
  83. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_datasources.py +0 -0
  84. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_diff.py +0 -0
  85. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_fields.py +0 -0
  86. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_joins.py +0 -0
  87. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_parser.py +0 -0
  88. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_published.py +0 -0
  89. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_relationships.py +0 -0
  90. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_rename.py +0 -0
  91. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_rename_mapping.py +0 -0
  92. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_sql.py +0 -0
  93. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_validators.py +0 -0
  94. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_version.py +0 -0
  95. {py_tbparse-0.3.0 → py_tbparse-0.4.4}/tests/test_xml.py +0 -0
@@ -14,14 +14,15 @@ type="join">` and 2020.2+ `<relationships>`), inferred relationships,
14
14
  dashboards/dashboard-sheets, `validate_relationships`, custom/initial SQL,
15
15
  and published-source detection. **Not yet ported** (v2): formatting,
16
16
  tooltips, colors, axes, sorts, dashboard layout/actions, analytics
17
- helpers (calc complexity, field usage, replication brief), and the
17
+ helpers (calc complexity, replication brief), and the
18
18
  Shiny-inspector equivalent.
19
19
 
20
20
  Also added, with no R equivalent (Python-native extras -- see "Non-R
21
21
  modules" below): Graphviz DOT export of the relationship graph
22
22
  (replaces the R package's igraph/ggraph-based plotting with a
23
23
  dependency-free alternative), workbook-to-workbook diff, folder/batch
24
- analysis across many workbooks, and Jupyter rich display.
24
+ analysis across many workbooks, field renaming, field usage, workbook
25
+ templates, and Jupyter rich display.
25
26
 
26
27
  ## Setup
27
28
 
@@ -120,11 +121,14 @@ beyond the R package's scope:
120
121
  | Python module | What it is |
121
122
  |---|---|
122
123
  | `_tables.py` | Name → `TwbParser`-accessor registry shared by `cli.py`, `webgui.py`, `diff.py`, and `batch.py`. Adding a new extractor to `TwbParser`? Add it here too so it's automatically available everywhere else. |
123
- | `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
124
- | `webgui.py` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
125
- | `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
124
+ | `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename`/`template` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
125
+ | `webgui.py` + `webui/` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. The server is `webgui.py`; the page is `webui/` (`index.html`, `tokens.css`, `themes.css`, `app.css`, `table.js`, `graph.js`, `app.js`). Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
126
+ | `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges); `graph_data()` is the same edges as plain data, which the GUI lays out and draws itself (`webui/graph.js`: layered layout, SVG, pan/zoom, keyboard). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
126
127
  | `diff.py` | `diff_tables()`/`diff_workbooks()`: row-level added/removed diff between two workbooks' same-named table, via `_tables.TABLE_SPECS`. No "changed" classification without a natural key — a changed row shows as one removed + one added row. |
127
- | `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` table in `_tables.py`. |
128
+ | `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `suggest_renames()` (the same for every kind of object: field, parameter, worksheet, dashboard, datasource, folder, hierarchy; adds a `kind` column), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` and `report-renames` tables in `_tables.py`. A worksheet/dashboard rename must rewrite every reference (`_SHEET_REFERENCES`); if you learn of another place Tableau writes a sheet name, add it there. |
129
+ | `report.py` | `workbook_report()`: the overview's report card (summary sentence, health checks, dashboards with their sheets) built from `validate_relationships`, `field_usage` and `usage.missing_references`. Every health item names the table and column filters that list exactly the rows it counted; a test keeps count and rows equal on all 200 corpus workbooks. |
130
+ | `usage.py` | `field_usage()`: for every field, the worksheets, dashboards and calculations that use it, followed through calculations, groups/sets and bins (the `field-usage` table). Python-native rather than a port of the R package's field-usage helper. |
131
+ | `templates.py` | Workbook templates. `make_template()` writes a `.twbx` with a `template.json` manifest (required fields from `usage.py`, parameters, connections; data, extracts and credentials stripped); `read_data()` describes new data (CSV, or a `.twb`/`.twbx`/`.tds`); `suggest_mapping()` pairs fields with columns (reuses `rename._match_key`, type checks against the field's *physical* type, `datatype-customized` fields keep their type); `apply_template()` replaces the datasource's connection, keeping every field's local name, sets parameters and stores `template-answers.json`. A CSV connection is written in the 2020.2+ object-model shape (both `_.fcp.ObjectModelEncapsulateLegacy` relations, object graph, table column, `object-id` per record) when the template uses it -- keep those in step if you touch one. Untested in Tableau itself. |
128
132
  | `batch.py` | `scan_folder()`: runs one table across every `.twb`/`.twbx` in a directory, concatenated with a `workbook` column. Skips (with a warning) any file that fails to load/extract rather than aborting the batch. |
129
133
 
130
134
  ## Porting conventions (read before adding/modifying a function)
@@ -174,6 +178,12 @@ beyond the R package's scope:
174
178
  synthetic XML snippet (see `tests/test_joins.py`,
175
179
  `tests/test_dashboards.py` for the pattern — many are lifted from the R
176
180
  functions' own `@examples` roxygen blocks).
181
+ - `tests/corpus/` is 200 real workbooks (permissive licences, pinned by blob sha in
182
+ `manifest.csv`, licence texts in `licenses/`). The files are gitignored; fetch with
183
+ `python scripts/fetch_corpus.py`. `tests/test_corpus.py` skips without them. Run it when you
184
+ change anything that reads workbook XML: it found a Unicode matching bug the hand-made
185
+ fixtures could not. Add to the corpus only from repositories whose licence permits
186
+ redistribution, and record the licence text.
177
187
  - `tests/conftest.py` provides `wenjie_xml`, `wenjie_path`,
178
188
  `zip_twbx_path` fixtures and an `xml_from_string()` helper. Tests import
179
189
  it with `from conftest import xml_from_string` (no `tests/__init__.py`,
@@ -194,21 +204,38 @@ string became a literal newline, splitting a JS string literal across
194
204
  two lines) killed the entire script, so no handlers bound and the UI was
195
205
  inert — with the whole suite green.
196
206
 
197
- So: **any change to `webgui.py`'s `_PAGE` needs a browser test**, in
207
+ So: **any change to the page (`py_tbparse/webui/`) needs a browser test**, in
208
+ The browser suite is split by concern, all sharing the fixtures in `tests/test_gui_browser.py`:
209
+ `test_gui_table.py` (windowing invariants, pipeline vs. a Python oracle, a seeded random walk;
210
+ `PYTBPARSE_WALK_SEEDS`/`PYTBPARSE_WALK_STEPS` widen it), `test_gui_a11y.py` (roles, keyboard, focus, rendered
211
+ contrast in both themes) and `test_gui_layout.py` (no sideways overflow from 320 px up, layout stability).
212
+ Design decisions the tests pin: only sorting uses a view transition (Chromium sends clicks to the page root while
213
+ one runs); column `MIN_WIDTH` is 80; per-table view settings are keyed by column name; disabled menu items use
214
+ `aria-disabled` so they stay focusable.
215
+
198
216
  `tests/test_gui_browser.py`, which runs the page in real headless
199
217
  Chromium via Playwright and fails on any uncaught JS error. The cheap
200
218
  structural guards in `test_webgui.py` (unterminated string literals,
201
219
  bracket balance) are a backstop, not a substitute.
202
220
 
203
- `_PAGE` is declared as `r"""..."""` (a **raw** string) specifically so
204
- this can't recur: without `r`, any backslash escape meant for the
205
- *browser* (`\n`, `\t`, a future `\'`) would need doubling in the Python
206
- source, and forgetting to double it silently corrupts the embedded JS
207
- instead of erroring. Keep it raw — write JS escapes the normal JS way
208
- (`'\n'`, not `'\\n'`).
209
-
210
- Any server-side value spliced into `_PAGE` (currently `TABLE_NAMES` and
211
- the preloaded workbook path) must go through `_json_for_script()`, not
221
+ The page is real files, not a Python string: `py_tbparse/webui/index.html`, `tokens.css` (the design
222
+ tokens), `app.css` and `app.js`. They are served by `webgui.py` from a fixed whitelist under
223
+ `/static/` (`_ASSETS`), so a request can never reach any other file; add a new asset to
224
+ `_ASSETS` and to the `webui/*` package-data (pyproject and MANIFEST.in) or it will not ship.
225
+ `index.html` is the only thing that gets server values: `_render_index()` replaces
226
+ `<!--APP_CONFIG-->` (now in `<head>`, so the saved theme is on `<html>` before the first paint) with the single
227
+ inline `<script>`. Keep it that way (one inline script plus the three external ones, `table.js`, `graph.js`, then
228
+ `app.js`); a test counts them. `populateTables();` must appear exactly once
229
+ in `app.js`, and the structural tests (unterminated string literals, bracket balance) read both served scripts.
230
+ Keep apostrophes out of JS strings and comments (the test counts quotes per line).
231
+
232
+ `POST /upload` (drag and drop, Open file) takes raw bytes with `Content-Type: application/octet-stream` and the name
233
+ in `X-Filename`: neither is CORS-safelisted, so another site cannot send it without a preflight. It streams to a
234
+ `mkdtemp` directory, refuses over `MAX_UPLOAD_BYTES` (413) and content that does not match the extension, and keeps
235
+ only one upload at a time. An uploaded workbook cannot "create beside the original" (409), only download.
236
+
237
+ Any server-side value spliced into the page (currently `TABLE_NAMES`, the preloaded workbook path
238
+ and the version, all in `_render_index()`'s config script) must go through `_json_for_script()`, not
212
239
  bare `json.dumps()`. `json.dumps` doesn't escape `/`, so a value
213
240
  containing the literal text `</script>` closes the script tag early in
214
241
  the browser's HTML parser — this is real, not theoretical, since the
@@ -232,6 +259,35 @@ leaves inputs empty and tests time out) and may crash rendering. Don't add
232
259
  startup, which makes collection fail for anyone who doesn't have it
233
260
  installed.
234
261
 
262
+ The table is windowed (`table.js`, class `VTable`): rows have a fixed height (`--row-h`) and only the rows
263
+ near the viewport exist in the page, so code and tests that read rows from the DOM see a window, not the table.
264
+ Use `aria-rowcount` for the size and the column menu's "Copy column values" to read a whole column (the tests do:
265
+ `_copy_column`). Never render every row; the old table needed over 4 s to sort 20,000 rows and could not draw
266
+ 50,000. Data work (search index, typed sort) is in `app.js`; each render records the User Timing measure
267
+ `py-tbparse:table`, which the speed tests read. The budgets (first paint of 1,000 rows 150 ms, filtering 50,000 rows
268
+ 100 ms) are enforced at twice the figure. Heavy tests skip below 1.5 GB of free memory (`_enough_memory`); this
269
+ sandbox has only 2 CPUs and 7.9 GB, and a 50,000-row DOM render has crashed it before.
270
+
271
+ The page's colours, spacing, type and motion are tokens in `webui/tokens.css`; use them rather than literals.
272
+ Colour themes are `webui/themes.css`: each theme is a complete colour set in light and dark, selected by
273
+ `<html data-theme data-mode>` (Auto is resolved to light/dark by the head script and kept in step by `app.js`;
274
+ dark colours live under `[data-mode="dark"]`, there is no `prefers-color-scheme` query). To add a theme, add its
275
+ two blocks there, its id to `webgui.THEMES` and its label to `THEME_LABELS` in `app.js`; the token test checks it.
276
+ `tests/test_webui_tokens.py` reads those files and fails if any text/background pair drops below WCAG AA
277
+ 4.5:1, if the stylesheet uses an undefined variable, or if text is faded with `opacity` (use `--faint`).
278
+ Motion durations come from the `--dur-*` tokens, which `prefers-reduced-motion` sets to instant; keep new
279
+ animations on those tokens.
280
+
281
+ `scripts/readme_screenshots.py` retakes the README's four images from the made-up `docs/demo/coffee-shop.twb`
282
+ (built by `scripts/make_demo_workbook.py`, which is the one place that workbook is defined); run both after a
283
+ visible change.
284
+
285
+ `scripts/gui_screenshots.py WORKBOOK OUT_DIR [--compare BASELINE_DIR]` captures nine GUI states (start,
286
+ overview and fields in light and dark, renames, graph, phone width) deterministically. Use it for GUI
287
+ refactors: a pure refactor must compare all-identical to the baseline taken before it; a redesign is expected to
288
+ differ, so review the new look and re-capture the baseline. The redesign plan and its decisions are in
289
+ `docs/ui-redesign-plan.md`.
290
+
235
291
  ## Commit / PR conventions
236
292
 
237
293
  Nothing project-specific beyond the harness defaults — see repo commit
@@ -5,3 +5,4 @@ recursive-include tests *.py
5
5
  include tests/fixtures/*.twb
6
6
  include tests/fixtures/*.twbx
7
7
  recursive-include docs *.png
8
+ recursive-include py_tbparse/webui *
@@ -0,0 +1,137 @@
1
+ Metadata-Version: 2.4
2
+ Name: py-tbparse
3
+ Version: 0.4.4
4
+ Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
5
+ Author: DDSNA
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/DDSNA/py-tbparse
8
+ Project-URL: Source, https://github.com/DDSNA/py-tbparse
9
+ Project-URL: Issues, https://github.com/DDSNA/py-tbparse/issues
10
+ Keywords: tableau,twb,twbx,workbook,parser,pandas
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
+ Classifier: Topic :: Office/Business
23
+ Requires-Python: >=3.9
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Requires-Dist: lxml>=4.9
27
+ Requires-Dist: pandas>=1.5
28
+ Provides-Extra: test
29
+ Requires-Dist: pytest>=7; extra == "test"
30
+ Provides-Extra: browser
31
+ Requires-Dist: playwright>=1.40; extra == "browser"
32
+ Provides-Extra: dev
33
+ Requires-Dist: build>=1.0; extra == "dev"
34
+ Requires-Dist: twine>=5.0; extra == "dev"
35
+ Dynamic: license-file
36
+
37
+ # py-tbparse
38
+
39
+ [![PyPI](https://img.shields.io/pypi/v/py-tbparse)](https://pypi.org/project/py-tbparse/)
40
+ [![Python](https://img.shields.io/pypi/pyversions/py-tbparse)](https://pypi.org/project/py-tbparse/)
41
+ [![CI](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml/badge.svg)](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml)
42
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE)
43
+
44
+ Read Tableau workbooks (`.twb` and `.twbx`) from Python. You get datasources, fields, calculated fields, joins, relationships and dashboards as pandas DataFrames. It is plain Python on top of `lxml` and `pandas`, so Tableau and R are not needed.
45
+
46
+ It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The command-line tool, the browser GUI, renaming, templates, workbook diffing and folder scanning are new here.
47
+
48
+ ![The overview report card for a small demo workbook](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-overview.png)
49
+
50
+ ## What it does
51
+
52
+ - Lists what is in a workbook: datasources, parameters, fields, calculations, joins, relationships, dashboards, custom SQL, published sources.
53
+ - Checks relationships and finds calculations that refer to fields the workbook does not have.
54
+ - Shows which worksheets, dashboards and calculations use each field.
55
+ - Suggests clean names for fields, sheets, dashboards and other objects, and writes a renamed copy.
56
+ - Turns a finished workbook into a template you can fill with other data.
57
+ - Compares two workbooks, scans a folder of them, and draws the data model as a graph.
58
+ - Works from Python, from the command line, or in a local browser page.
59
+
60
+ ## Install
61
+
62
+ ```bash
63
+ pip install py-tbparse
64
+ ```
65
+
66
+ Python 3.9 or newer. The package, the import and both commands all use the same name: `pip install py-tbparse`, `import py_tbparse`, `py-tbparse`, `py-tbparse-gui`. Releases up to 0.2.0 used `twbparser_py` and `twbparser`; those names are gone.
67
+
68
+ ## Quick start
69
+
70
+ From Python:
71
+
72
+ ```python
73
+ from py_tbparse import TwbParser
74
+
75
+ p = TwbParser("workbook.twb") # or a .twbx
76
+ p.get_overview() # counts of everything
77
+ p.get_calculated_fields() # each calculation and its formula
78
+ p.get_relationships() # how the tables connect
79
+ ```
80
+
81
+ Every getter returns a DataFrame. The others are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql`, `get_published_refs` and `get_field_usage`. `get_relationship_graph_dot()` returns the data model as Graphviz text and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
82
+
83
+ From the command line:
84
+
85
+ ```bash
86
+ py-tbparse workbook.twb # overview
87
+ py-tbparse workbook.twb calculated-fields
88
+ py-tbparse workbook.twb fields --format csv -o fields.csv
89
+ py-tbparse workbook.twb validate # exit code 2 if it finds a problem
90
+ ```
91
+
92
+ In the browser:
93
+
94
+ ```bash
95
+ py-tbparse-gui workbook.twb
96
+ ```
97
+
98
+ Across workbooks:
99
+
100
+ ```python
101
+ from py_tbparse import TwbParser, diff_workbooks, scan_folder
102
+
103
+ diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
104
+ scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
105
+ ```
106
+
107
+ ## Guides
108
+
109
+ - [Command line](https://github.com/DDSNA/py-tbparse/blob/main/docs/cli.md): every table, the `diff`, `batch`, `rename` and `template` commands, and their options.
110
+ - [Browser GUI](https://github.com/DDSNA/py-tbparse/blob/main/docs/gui.md): opening files, the table, the overview, the graph, themes.
111
+ - [Renaming](https://github.com/DDSNA/py-tbparse/blob/main/docs/renaming.md): clean names after a datasource switch, editing the suggestions, what will stay broken.
112
+ - [Templates](https://github.com/DDSNA/py-tbparse/blob/main/docs/templates.md): make a template from a workbook and apply it to new data.
113
+ - [`.twbx` files](https://github.com/DDSNA/py-tbparse/blob/main/docs/twbx.md): they are read straight from the zip; how to extract the contents.
114
+ - [Development](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md): running the tests, the workbook corpus, the browser tests.
115
+
116
+ ## Limits
117
+
118
+ - **Nothing the tool writes has been opened in Tableau yet.** That covers renamed workbooks and workbooks made from templates. The XML follows what Tableau writes, but open one on a copy and check it before relying on it. The original file is never modified or overwritten.
119
+ - **Only part of the R package is ported.** Missing: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, calculation complexity, the replication brief and the Shiny inspector. The GUI covers some of what the inspector did.
120
+ - Where the R version has a bug, this one does not copy it: joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
121
+ - The GUI is meant to run on your own machine for one person. It refuses requests that come from other websites, but there is no login, so do not put it on a shared network.
122
+
123
+ ## Why a Tableau parser
124
+
125
+ A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python is not hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
126
+
127
+ ## Contributing
128
+
129
+ Issues and pull requests are welcome at [github.com/DDSNA/py-tbparse](https://github.com/DDSNA/py-tbparse). The test setup is in the [development guide](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md). `AGENTS.md` describes the code layout and the rules for porting a function from the R package.
130
+
131
+ ## Credit
132
+
133
+ Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
134
+
135
+ ## License
136
+
137
+ MIT, see [LICENSE](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE).
@@ -0,0 +1,101 @@
1
+ # py-tbparse
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/py-tbparse)](https://pypi.org/project/py-tbparse/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/py-tbparse)](https://pypi.org/project/py-tbparse/)
5
+ [![CI](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml/badge.svg)](https://github.com/DDSNA/py-tbparse/actions/workflows/ci.yml)
6
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE)
7
+
8
+ Read Tableau workbooks (`.twb` and `.twbx`) from Python. You get datasources, fields, calculated fields, joins, relationships and dashboards as pandas DataFrames. It is plain Python on top of `lxml` and `pandas`, so Tableau and R are not needed.
9
+
10
+ It began as a port of PrigasG's R package [twbparser](https://github.com/PrigasG/twbparser). The command-line tool, the browser GUI, renaming, templates, workbook diffing and folder scanning are new here.
11
+
12
+ ![The overview report card for a small demo workbook](https://raw.githubusercontent.com/DDSNA/py-tbparse/main/docs/gui-overview.png)
13
+
14
+ ## What it does
15
+
16
+ - Lists what is in a workbook: datasources, parameters, fields, calculations, joins, relationships, dashboards, custom SQL, published sources.
17
+ - Checks relationships and finds calculations that refer to fields the workbook does not have.
18
+ - Shows which worksheets, dashboards and calculations use each field.
19
+ - Suggests clean names for fields, sheets, dashboards and other objects, and writes a renamed copy.
20
+ - Turns a finished workbook into a template you can fill with other data.
21
+ - Compares two workbooks, scans a folder of them, and draws the data model as a graph.
22
+ - Works from Python, from the command line, or in a local browser page.
23
+
24
+ ## Install
25
+
26
+ ```bash
27
+ pip install py-tbparse
28
+ ```
29
+
30
+ Python 3.9 or newer. The package, the import and both commands all use the same name: `pip install py-tbparse`, `import py_tbparse`, `py-tbparse`, `py-tbparse-gui`. Releases up to 0.2.0 used `twbparser_py` and `twbparser`; those names are gone.
31
+
32
+ ## Quick start
33
+
34
+ From Python:
35
+
36
+ ```python
37
+ from py_tbparse import TwbParser
38
+
39
+ p = TwbParser("workbook.twb") # or a .twbx
40
+ p.get_overview() # counts of everything
41
+ p.get_calculated_fields() # each calculation and its formula
42
+ p.get_relationships() # how the tables connect
43
+ ```
44
+
45
+ Every getter returns a DataFrame. The others are `get_datasources`, `get_parameters`, `get_fields`, `get_raw_fields`, `get_joins`, `get_relations`, `get_inferred_relationships`, `get_dashboards`, `get_dashboard_sheets`, `get_custom_sql`, `get_initial_sql`, `get_published_refs` and `get_field_usage`. `get_relationship_graph_dot()` returns the data model as Graphviz text and `validate()` looks for problems in the relationships. In Jupyter, a bare `p` shows the overview.
46
+
47
+ From the command line:
48
+
49
+ ```bash
50
+ py-tbparse workbook.twb # overview
51
+ py-tbparse workbook.twb calculated-fields
52
+ py-tbparse workbook.twb fields --format csv -o fields.csv
53
+ py-tbparse workbook.twb validate # exit code 2 if it finds a problem
54
+ ```
55
+
56
+ In the browser:
57
+
58
+ ```bash
59
+ py-tbparse-gui workbook.twb
60
+ ```
61
+
62
+ Across workbooks:
63
+
64
+ ```python
65
+ from py_tbparse import TwbParser, diff_workbooks, scan_folder
66
+
67
+ diff_workbooks(TwbParser("v1.twb"), TwbParser("v2.twb"), table="datasources")
68
+ scan_folder("./workbooks", table="datasources") # one table, every workbook in the folder
69
+ ```
70
+
71
+ ## Guides
72
+
73
+ - [Command line](https://github.com/DDSNA/py-tbparse/blob/main/docs/cli.md): every table, the `diff`, `batch`, `rename` and `template` commands, and their options.
74
+ - [Browser GUI](https://github.com/DDSNA/py-tbparse/blob/main/docs/gui.md): opening files, the table, the overview, the graph, themes.
75
+ - [Renaming](https://github.com/DDSNA/py-tbparse/blob/main/docs/renaming.md): clean names after a datasource switch, editing the suggestions, what will stay broken.
76
+ - [Templates](https://github.com/DDSNA/py-tbparse/blob/main/docs/templates.md): make a template from a workbook and apply it to new data.
77
+ - [`.twbx` files](https://github.com/DDSNA/py-tbparse/blob/main/docs/twbx.md): they are read straight from the zip; how to extract the contents.
78
+ - [Development](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md): running the tests, the workbook corpus, the browser tests.
79
+
80
+ ## Limits
81
+
82
+ - **Nothing the tool writes has been opened in Tableau yet.** That covers renamed workbooks and workbooks made from templates. The XML follows what Tableau writes, but open one on a copy and check it before relying on it. The original file is never modified or overwritten.
83
+ - **Only part of the R package is ported.** Missing: formatting, tooltips, colors, axes and sorts, dashboard layout and actions, calculation complexity, the replication brief and the Shiny inspector. The GUI covers some of what the inspector did.
84
+ - Where the R version has a bug, this one does not copy it: joins and relationships on more than one key, nested joins, the include-parameters option, and calculations with brackets inside brackets.
85
+ - The GUI is meant to run on your own machine for one person. It refuses requests that come from other websites, but there is no login, so do not put it on a shared network.
86
+
87
+ ## Why a Tableau parser
88
+
89
+ A `.twb` is XML and a `.twbx` is a zip containing one, so reading them from Python is not hard. Power BI's `.pbix` is a binary format built on a proprietary storage engine, and getting into it from code took reverse-engineering projects like [PBIXRay](https://github.com/Hugoberry/pbixray) and [pbi-tools](https://github.com/pbi-tools/pbi-tools). On the server side it goes the other way: Tableau's own Python tooling ([tableauserverclient](https://pypi.org/project/tableauserverclient/) and `tabcmd`) is more mature and more open than what Microsoft has for the Power BI REST API.
90
+
91
+ ## Contributing
92
+
93
+ Issues and pull requests are welcome at [github.com/DDSNA/py-tbparse](https://github.com/DDSNA/py-tbparse). The test setup is in the [development guide](https://github.com/DDSNA/py-tbparse/blob/main/docs/development.md). `AGENTS.md` describes the code layout and the rules for porting a function from the R package.
94
+
95
+ ## Credit
96
+
97
+ Based on [twbparser](https://github.com/PrigasG/twbparser) by George Arthur, MIT licensed.
98
+
99
+ ## License
100
+
101
+ MIT, see [LICENSE](https://github.com/DDSNA/py-tbparse/blob/main/LICENSE).
Binary file
Binary file
Binary file
@@ -10,7 +10,7 @@ from .dashboards import dashboard_sheets, list_dashboards
10
10
  from .datasources import extract_datasource_details, extract_named_connections, extract_parameters
11
11
  from .diff import diff_tables, diff_workbooks
12
12
  from .fields import extract_columns_with_table_source, infer_implicit_relationships
13
- from .graph import to_dot
13
+ from .graph import graph_data, to_dot
14
14
  from .joins import extract_joins
15
15
  from .parser import TwbParser
16
16
  from .published import extract_published_refs
@@ -21,8 +21,20 @@ from .rename import (
21
21
  load_rename_mapping,
22
22
  normalize_name,
23
23
  suggest_field_renames,
24
+ suggest_renames,
24
25
  )
25
26
  from .sql import extract_custom_sql, extract_initial_sql
27
+ from .templates import (
28
+ Template,
29
+ TemplateError,
30
+ apply_template,
31
+ load_mapping,
32
+ load_template,
33
+ make_template,
34
+ read_data,
35
+ suggest_mapping,
36
+ )
37
+ from .usage import field_usage
26
38
  from .validators import validate_relationships
27
39
  from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
28
40
 
@@ -39,6 +51,7 @@ __all__ = [
39
51
  "infer_implicit_relationships",
40
52
  "extract_joins",
41
53
  "to_dot",
54
+ "graph_data",
42
55
  "extract_relations",
43
56
  "extract_relationships",
44
57
  "extract_custom_sql",
@@ -56,6 +69,16 @@ __all__ = [
56
69
  "load_rename_mapping",
57
70
  "normalize_name",
58
71
  "suggest_field_renames",
72
+ "suggest_renames",
73
+ "field_usage",
74
+ "Template",
75
+ "TemplateError",
76
+ "make_template",
77
+ "load_template",
78
+ "read_data",
79
+ "suggest_mapping",
80
+ "load_mapping",
81
+ "apply_template",
59
82
  ]
60
83
 
61
84
  try:
@@ -39,12 +39,28 @@ def _calculated_fields(p: TwbParser, include_parameters: bool = False, **_kw) ->
39
39
 
40
40
 
41
41
  def _field_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
42
- datasource=None, **_kw) -> pd.DataFrame:
42
+ datasource=None, kinds=None, **_kw) -> pd.DataFrame:
43
+ if kinds: # the GUI's "everything in the report" switch
44
+ return p.get_renames(
45
+ reference=reference, style=style, only_changed=only_changed, datasource=datasource, kinds=kinds
46
+ )
43
47
  return p.get_field_renames(
44
48
  reference=reference, style=style, only_changed=only_changed, datasource=datasource
45
49
  )
46
50
 
47
51
 
52
+ def _report_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
53
+ datasource=None, **_kw) -> pd.DataFrame:
54
+ return p.get_renames(reference=reference, style=style, only_changed=only_changed, datasource=datasource)
55
+
56
+
57
+ def _field_usage(p: TwbParser, **_kw) -> pd.DataFrame:
58
+ df = p.get_field_usage()
59
+ for col in ("sheets", "dashboards", "calculations"):
60
+ df[col] = df[col].map("; ".join)
61
+ return df
62
+
63
+
48
64
  def _joins(p: TwbParser, **_kw) -> pd.DataFrame:
49
65
  return p.get_joins()
50
66
 
@@ -77,6 +93,10 @@ def _initial_sql(p: TwbParser, **_kw) -> pd.DataFrame:
77
93
  return p.get_initial_sql()
78
94
 
79
95
 
96
+ def _missing_references(p: TwbParser, **_kw) -> pd.DataFrame:
97
+ return p.get_missing_references()
98
+
99
+
80
100
  def _published_refs(p: TwbParser, **_kw) -> pd.DataFrame:
81
101
  return p.get_published_refs()
82
102
 
@@ -91,6 +111,9 @@ TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
91
111
  "raw-fields": _raw_fields,
92
112
  "calculated-fields": _calculated_fields,
93
113
  "field-renames": _field_renames,
114
+ "report-renames": _report_renames,
115
+ "field-usage": _field_usage,
116
+ "missing-references": _missing_references,
94
117
  "joins": _joins,
95
118
  "relations": _relations,
96
119
  "relationships": _relationships,