py-tbparse 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/AGENTS.md +13 -4
  2. {py_tbparse-0.3.0/py_tbparse.egg-info → py_tbparse-0.4.0}/PKG-INFO +57 -3
  3. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/README.md +56 -2
  4. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/__init__.py +22 -0
  5. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/_tables.py +19 -1
  6. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/cli.py +124 -1
  7. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/parser.py +15 -0
  8. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/rename.py +272 -33
  9. py_tbparse-0.4.0/py_tbparse/templates.py +855 -0
  10. py_tbparse-0.4.0/py_tbparse/usage.py +166 -0
  11. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/webgui.py +16 -3
  12. {py_tbparse-0.3.0 → py_tbparse-0.4.0/py_tbparse.egg-info}/PKG-INFO +57 -3
  13. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/SOURCES.txt +5 -0
  14. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/pyproject.toml +1 -1
  15. py_tbparse-0.4.0/tests/test_corpus.py +84 -0
  16. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_gui_browser.py +23 -0
  17. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_public_workbooks.py +18 -0
  18. py_tbparse-0.4.0/tests/test_rename_all.py +215 -0
  19. py_tbparse-0.4.0/tests/test_templates.py +415 -0
  20. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_webgui.py +25 -0
  21. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/LICENSE +0 -0
  22. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/MANIFEST.in +0 -0
  23. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/docs/gui-field-renames.png +0 -0
  24. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/docs/gui-overview.png +0 -0
  25. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/_clean.py +0 -0
  26. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/_xml.py +0 -0
  27. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/batch.py +0 -0
  28. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/calculated_fields.py +0 -0
  29. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/dashboards.py +0 -0
  30. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/datasources.py +0 -0
  31. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/diff.py +0 -0
  32. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/fields.py +0 -0
  33. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/graph.py +0 -0
  34. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/joins.py +0 -0
  35. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/published.py +0 -0
  36. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/py.typed +0 -0
  37. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/relationships.py +0 -0
  38. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/sql.py +0 -0
  39. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse/validators.py +0 -0
  40. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/dependency_links.txt +0 -0
  41. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/entry_points.txt +0 -0
  42. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/requires.txt +0 -0
  43. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/py_tbparse.egg-info/top_level.txt +0 -0
  44. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/setup.cfg +0 -0
  45. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/conftest.py +0 -0
  46. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/fixtures/test_for_wenjie.twb +0 -0
  47. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/fixtures/test_for_zip.twbx +0 -0
  48. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_batch.py +0 -0
  49. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_calculated_fields.py +0 -0
  50. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_clean.py +0 -0
  51. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_cli.py +0 -0
  52. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_dashboards.py +0 -0
  53. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_datasources.py +0 -0
  54. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_diff.py +0 -0
  55. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_fields.py +0 -0
  56. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_graph.py +0 -0
  57. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_joins.py +0 -0
  58. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_parser.py +0 -0
  59. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_published.py +0 -0
  60. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_relationships.py +0 -0
  61. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_rename.py +0 -0
  62. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_rename_mapping.py +0 -0
  63. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_sql.py +0 -0
  64. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_validators.py +0 -0
  65. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_version.py +0 -0
  66. {py_tbparse-0.3.0 → py_tbparse-0.4.0}/tests/test_xml.py +0 -0
@@ -14,14 +14,15 @@ type="join">` and 2020.2+ `<relationships>`), inferred relationships,
14
14
  dashboards/dashboard-sheets, `validate_relationships`, custom/initial SQL,
15
15
  and published-source detection. **Not yet ported** (v2): formatting,
16
16
  tooltips, colors, axes, sorts, dashboard layout/actions, analytics
17
- helpers (calc complexity, field usage, replication brief), and the
17
+ helpers (calc complexity, replication brief), and the
18
18
  Shiny-inspector equivalent.
19
19
 
20
20
  Also added, with no R equivalent (Python-native extras -- see "Non-R
21
21
  modules" below): Graphviz DOT export of the relationship graph
22
22
  (replaces the R package's igraph/ggraph-based plotting with a
23
23
  dependency-free alternative), workbook-to-workbook diff, folder/batch
24
- analysis across many workbooks, and Jupyter rich display.
24
+ analysis across many workbooks, field renaming, field usage, workbook
25
+ templates, and Jupyter rich display.
25
26
 
26
27
  ## Setup
27
28
 
@@ -120,11 +121,13 @@ beyond the R package's scope:
120
121
  | Python module | What it is |
121
122
  |---|---|
122
123
  | `_tables.py` | Name → `TwbParser`-accessor registry shared by `cli.py`, `webgui.py`, `diff.py`, and `batch.py`. Adding a new extractor to `TwbParser`? Add it here too so it's automatically available everywhere else. |
123
- | `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
124
+ | `cli.py` | `py-tbparse` command-line entry point, plus the `diff`/`batch`/`rename`/`template` subcommands (dispatched on `sys.argv[1]` before the normal single-workbook argparse parser runs) |
124
125
  | `webgui.py` | `py-tbparse-gui`: stdlib-only (`http.server` + vanilla JS) local browser GUI, no GUI toolkit dependency. Loosely fills the role of the R package's `run_twbparser_app`/Shiny inspector. |
125
126
  | `graph.py` | `to_dot()`: Graphviz DOT export of joins/relationships (+ optional inferred, as dashed edges). Replaces the R package's igraph/ggraph-based `plot_dependency_graph`/`plot_relationship_graph` with a dependency-free text format any Graphviz-compatible tool can render. |
126
127
  | `diff.py` | `diff_tables()`/`diff_workbooks()`: row-level added/removed diff between two workbooks' same-named table, via `_tables.TABLE_SPECS`. No "changed" classification without a natural key — a changed row shows as one removed + one added row. |
127
- | `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` table in `_tables.py`. |
128
+ | `rename.py` | `suggest_field_renames()` (clean-name suggestions, optionally matched against a "before" reference), `suggest_renames()` (the same for every kind of object: field, parameter, worksheet, dashboard, datasource, folder, hierarchy; adds a `kind` column), `load_rename_mapping()` (read an edited CSV back), `compare_field_schemas()` (fields with no counterpart across a datasource switch) and `apply_field_renames()`/`build_renamed_workbook()` (write a copy with captions set; never overwrites). Also the `field-renames` and `report-renames` tables in `_tables.py`. A worksheet/dashboard rename must rewrite every reference (`_SHEET_REFERENCES`); if you learn of another place Tableau writes a sheet name, add it there. |
129
+ | `usage.py` | `field_usage()`: for every field, the worksheets, dashboards and calculations that use it, followed through calculations, groups/sets and bins (the `field-usage` table). Python-native rather than a port of the R package's field-usage helper. |
130
+ | `templates.py` | Workbook templates. `make_template()` writes a `.twbx` with a `template.json` manifest (required fields from `usage.py`, parameters, connections; data, extracts and credentials stripped); `read_data()` describes new data (CSV, or a `.twb`/`.twbx`/`.tds`); `suggest_mapping()` pairs fields with columns (reuses `rename._match_key`, type checks against the field's *physical* type, `datatype-customized` fields keep their type); `apply_template()` replaces the datasource's connection, keeping every field's local name, sets parameters and stores `template-answers.json`. A CSV connection is written in the 2020.2+ object-model shape (both `_.fcp.ObjectModelEncapsulateLegacy` relations, object graph, table column, `object-id` per record) when the template uses it -- keep those in step if you touch one. Untested in Tableau itself. |
128
131
  | `batch.py` | `scan_folder()`: runs one table across every `.twb`/`.twbx` in a directory, concatenated with a `workbook` column. Skips (with a warning) any file that fails to load/extract rather than aborting the batch. |
129
132
 
130
133
  ## Porting conventions (read before adding/modifying a function)
@@ -174,6 +177,12 @@ beyond the R package's scope:
174
177
  synthetic XML snippet (see `tests/test_joins.py`,
175
178
  `tests/test_dashboards.py` for the pattern — many are lifted from the R
176
179
  functions' own `@examples` roxygen blocks).
180
+ - `tests/corpus/` is 200 real workbooks (permissive licences, pinned by blob sha in
181
+ `manifest.csv`, licence texts in `licenses/`). The files are gitignored; fetch with
182
+ `python scripts/fetch_corpus.py`. `tests/test_corpus.py` skips without them. Run it when you
183
+ change anything that reads workbook XML: it found a Unicode matching bug the hand-made
184
+ fixtures could not. Add to the corpus only from repositories whose licence permits
185
+ redistribution, and record the licence text.
177
186
  - `tests/conftest.py` provides `wenjie_xml`, `wenjie_path`,
178
187
  `zip_twbx_path` fixtures and an `xml_from_string()` helper. Tests import
179
188
  it with `from conftest import xml_from_string` (no `tests/__init__.py`,
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: py-tbparse
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: Native Python port of the twbparser R package: parse Tableau .twb/.twbx workbooks into pandas DataFrames.
5
5
  Author: DDSNA
6
6
  License: MIT
@@ -101,6 +101,24 @@ With a `reference` (a workbook, a fields table or a plain list of names), fields
101
101
 
102
102
  `p.get_field_renames()` does the same from a parser.
103
103
 
104
+ **Rename everything in the report, not just fields.** Pass `kinds` (or `--all` / `--kinds` on the command line) and worksheets, dashboards, datasources, parameters, folders and hierarchies are covered too:
105
+
106
+ ```python
107
+ from py_tbparse import TwbParser, suggest_renames, apply_field_renames
108
+
109
+ p = TwbParser("report.twb")
110
+ suggest_renames(p, only_changed=True) # kind, datasource, name, current, suggested, ...
111
+ suggest_renames(p, kinds=["worksheet", "dashboard"]) # just the sheets
112
+ apply_field_renames(p, kinds="all") # writes report_renamed.twb
113
+ ```
114
+
115
+ ```bash
116
+ py-tbparse rename report.twb --all --only-changed
117
+ py-tbparse rename report.twb --kinds worksheet,dashboard --write-workbook
118
+ ```
119
+
120
+ Each kind is renamed the way Tableau does it: fields, parameters and datasources get a caption (their internal names stay, so formulas and sheets keep working); a worksheet or dashboard is renamed in every place its name is written (the sheet, its window and thumbnail, the zones of dashboards that show it, actions, story points); a folder or hierarchy gets its new name. Worksheets and dashboards share one namespace, as they do in Tableau, so two of them never end up with the same name. A `reference` workbook lends its spelling to objects of the same kind. Datasources that Tableau named itself (`federated.0grg...`) and nobody captioned are left out. The `kind` column also appears in the CSV, so **Edit the suggestions yourself** works for sheets too. The `report-renames` table lists all of it, and the GUI's Field renames view has an "Everything in the report" switch. As with fields, none of this has been opened in Tableau itself.
121
+
104
122
  **Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
105
123
 
106
124
  ```bash
@@ -116,6 +134,37 @@ To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...
116
134
 
117
135
  **When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
118
136
 
137
+ ### Templates
138
+
139
+ Turn a finished workbook into a template, then make new workbooks from it with other data. Every sheet, dashboard, calculation and format comes along; only the data changes. The idea comes from Tableau's Accelerators and Power BI's `.pbit` files.
140
+
141
+ ```bash
142
+ py-tbparse template make sales.twbx # writes sales.template.twbx
143
+ py-tbparse template show sales.template.twbx # the fields it needs, and its parameters
144
+ py-tbparse template apply sales.template.twbx --data q3.csv # suggested mapping + what would break; writes nothing
145
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping-out map.csv # save the mapping to edit
146
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping map.csv -p "Top N=10" --write
147
+ ```
148
+
149
+ ```python
150
+ from py_tbparse import make_template, load_template, read_data, suggest_mapping, apply_template
151
+
152
+ t = load_template(make_template("sales.twbx"))
153
+ data = read_data("q3.csv") # or a .twb / .twbx / .tds already connected to the new data
154
+ suggest_mapping(t, data) # field, required, used_by, mapped_to, status, ...
155
+ apply_template(t, data, params={"Top N": "10"}) # writes sales_q3.twbx
156
+ ```
157
+
158
+ **What a template is.** An ordinary `.twbx` (Tableau still opens it) with a `template.json` manifest inside. The manifest lists the fields the workbook takes from its data, marking a field `required` when a sheet uses it, directly or through calculations, groups and sets. It also lists the parameters and where the data came from. Extracts, packaged data and cached query results are left out (`--keep-data` keeps them as sample data), and user names and passwords are blanked.
159
+
160
+ **Mapping.** Each required field is matched to a column of the new data by name, ignoring case and separators (`ORDER_DATE` → `Order Date`), with close spellings accepted above `--cutoff`. Types are checked like Tableau's Accelerator mapper: a text column is never offered for a number or a date, while integer vs decimal and date vs date-time map with a warning. A field whose type the author changed in Tableau keeps that type, and Tableau converts the column. Before anything is written you see which sheets would break for each field left without a column; writing then needs `--allow-missing`. Edit the mapping as a CSV, as with renames.
161
+
162
+ **What gets written.** The template's connection is replaced by one to the new data (a CSV file, or the connection of the workbook / `.tds` you pass). Every field keeps the local name its sheets and formulas use; only the physical column behind it changes. Parameter values are set with `-p NAME=VALUE`, checked against the parameter's type and its list of allowed values. The output (`<template>_<data>.twbx` beside the template, never overwritten) also stores `template-answers.json`: which template, data, mapping and parameters made it, so it can be re-made or checked later.
163
+
164
+ **Limits.** A CSV feeds one table. A template whose datasource joins several tables needs a workbook or `.tds` as its data, so the joins come along. Excel files are not read directly yet; save as CSV or pass a workbook connected to the sheet. As with renames, I have not opened the generated workbooks in Tableau itself, so check one before relying on it.
165
+
166
+ `p.get_field_usage()` (or `field_usage(p)`, the `field-usage` table) is the analysis behind `required`: for every field, the sheets, dashboards and calculations that use it.
167
+
119
168
  ### `.twbx` files
120
169
 
121
170
  A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
@@ -141,11 +190,14 @@ py-tbparse diff old.twb new.twb datasources
141
190
  py-tbparse batch ./workbooks datasources
142
191
  py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
143
192
  py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
193
+ py-tbparse rename report.twb --all --write-workbook # sheets, dashboards, datasources, ... too
194
+ py-tbparse template make sales.twbx # see Templates above
195
+ py-tbparse template apply sales.template.twbx --data q3.csv --write
144
196
  ```
145
197
 
146
- Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`.
198
+ Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`, `field-usage`, `field-renames`, `report-renames`.
147
199
 
148
- `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
200
+ `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--all`, `--kinds`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
149
201
 
150
202
  ## GUI
151
203
 
@@ -172,6 +224,8 @@ pytest
172
224
 
173
225
  The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
174
226
 
227
+ `tests/corpus/` lists 200 more real workbooks from public repositories with MIT, Apache-2.0, ISC or CC0 licences, for integration tests and as examples (manifest, licence texts and where each file came from are in its README). The files themselves are not in git (about 26 MB): run `python scripts/fetch_corpus.py` to download them, checked against the manifest. `tests/test_corpus.py` then runs every feature over all of them; it skips when they are not fetched.
228
+
175
229
  The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
176
230
 
177
231
  ```bash
@@ -65,6 +65,24 @@ With a `reference` (a workbook, a fields table or a plain list of names), fields
65
65
 
66
66
  `p.get_field_renames()` does the same from a parser.
67
67
 
68
+ **Rename everything in the report, not just fields.** Pass `kinds` (or `--all` / `--kinds` on the command line) and worksheets, dashboards, datasources, parameters, folders and hierarchies are covered too:
69
+
70
+ ```python
71
+ from py_tbparse import TwbParser, suggest_renames, apply_field_renames
72
+
73
+ p = TwbParser("report.twb")
74
+ suggest_renames(p, only_changed=True) # kind, datasource, name, current, suggested, ...
75
+ suggest_renames(p, kinds=["worksheet", "dashboard"]) # just the sheets
76
+ apply_field_renames(p, kinds="all") # writes report_renamed.twb
77
+ ```
78
+
79
+ ```bash
80
+ py-tbparse rename report.twb --all --only-changed
81
+ py-tbparse rename report.twb --kinds worksheet,dashboard --write-workbook
82
+ ```
83
+
84
+ Each kind is renamed the way Tableau does it: fields, parameters and datasources get a caption (their internal names stay, so formulas and sheets keep working); a worksheet or dashboard is renamed in every place its name is written (the sheet, its window and thumbnail, the zones of dashboards that show it, actions, story points); a folder or hierarchy gets its new name. Worksheets and dashboards share one namespace, as they do in Tableau, so two of them never end up with the same name. A `reference` workbook lends its spelling to objects of the same kind. Datasources that Tableau named itself (`federated.0grg...`) and nobody captioned are left out. The `kind` column also appears in the CSV, so **Edit the suggestions yourself** works for sheets too. The `report-renames` table lists all of it, and the GUI's Field renames view has an "Everything in the report" switch. As with fields, none of this has been opened in Tableau itself.
85
+
68
86
  **Edit the suggestions yourself.** Export them, change the `suggested` column in a spreadsheet (blank means leave the field alone), then apply your version to a copy of the workbook:
69
87
 
70
88
  ```bash
@@ -80,6 +98,37 @@ To save the result, `p.write_renamed_workbook()` (or `apply_field_renames(p, ...
80
98
 
81
99
  **When to run it.** Add the new datasource to a *copy* of the workbook first, then run this with the old workbook as `reference` and `datasource=` set to the new source, so only its fields are renamed. Open the fixed copy and use Replace Data Source; fields with matching names should re-link on their own. It also works after references have already broken, but it only fixes names: sheets that point at missing fields stay broken until you replace the source again. (Check this on a copy first; I have not tested the re-linking in Tableau itself.)
82
100
 
101
+ ### Templates
102
+
103
+ Turn a finished workbook into a template, then make new workbooks from it with other data. Every sheet, dashboard, calculation and format comes along; only the data changes. The idea comes from Tableau's Accelerators and Power BI's `.pbit` files.
104
+
105
+ ```bash
106
+ py-tbparse template make sales.twbx # writes sales.template.twbx
107
+ py-tbparse template show sales.template.twbx # the fields it needs, and its parameters
108
+ py-tbparse template apply sales.template.twbx --data q3.csv # suggested mapping + what would break; writes nothing
109
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping-out map.csv # save the mapping to edit
110
+ py-tbparse template apply sales.template.twbx --data q3.csv --mapping map.csv -p "Top N=10" --write
111
+ ```
112
+
113
+ ```python
114
+ from py_tbparse import make_template, load_template, read_data, suggest_mapping, apply_template
115
+
116
+ t = load_template(make_template("sales.twbx"))
117
+ data = read_data("q3.csv") # or a .twb / .twbx / .tds already connected to the new data
118
+ suggest_mapping(t, data) # field, required, used_by, mapped_to, status, ...
119
+ apply_template(t, data, params={"Top N": "10"}) # writes sales_q3.twbx
120
+ ```
121
+
122
+ **What a template is.** An ordinary `.twbx` (Tableau still opens it) with a `template.json` manifest inside. The manifest lists the fields the workbook takes from its data, marking a field `required` when a sheet uses it, directly or through calculations, groups and sets. It also lists the parameters and where the data came from. Extracts, packaged data and cached query results are left out (`--keep-data` keeps them as sample data), and user names and passwords are blanked.
123
+
124
+ **Mapping.** Each required field is matched to a column of the new data by name, ignoring case and separators (`ORDER_DATE` → `Order Date`), with close spellings accepted above `--cutoff`. Types are checked like Tableau's Accelerator mapper: a text column is never offered for a number or a date, while integer vs decimal and date vs date-time map with a warning. A field whose type the author changed in Tableau keeps that type, and Tableau converts the column. Before anything is written you see which sheets would break for each field left without a column; writing then needs `--allow-missing`. Edit the mapping as a CSV, as with renames.
125
+
126
+ **What gets written.** The template's connection is replaced by one to the new data (a CSV file, or the connection of the workbook / `.tds` you pass). Every field keeps the local name its sheets and formulas use; only the physical column behind it changes. Parameter values are set with `-p NAME=VALUE`, checked against the parameter's type and its list of allowed values. The output (`<template>_<data>.twbx` beside the template, never overwritten) also stores `template-answers.json`: which template, data, mapping and parameters made it, so it can be re-made or checked later.
127
+
128
+ **Limits.** A CSV feeds one table. A template whose datasource joins several tables needs a workbook or `.tds` as its data, so the joins come along. Excel files are not read directly yet; save as CSV or pass a workbook connected to the sheet. As with renames, I have not opened the generated workbooks in Tableau itself, so check one before relying on it.
129
+
130
+ `p.get_field_usage()` (or `field_usage(p)`, the `field-usage` table) is the analysis behind `required`: for every field, the sheets, dashboards and calculations that use it.
131
+
83
132
  ### `.twbx` files
84
133
 
85
134
  A `.twbx` is read directly from the zip and nothing gets written to disk. That means `p.twbx_dir` is `None`, and `p.path` is a made-up `<file>.twbx/<name>.twb` that you can't open. If you want the files out, extract them yourself:
@@ -105,11 +154,14 @@ py-tbparse diff old.twb new.twb datasources
105
154
  py-tbparse batch ./workbooks datasources
106
155
  py-tbparse rename new.twb --reference old.twb --only-changed # suggested clean field names
107
156
  py-tbparse rename new.twb -r old.twb --datasource federated.abc123 --write-workbook # and save new_renamed.twb
157
+ py-tbparse rename report.twb --all --write-workbook # sheets, dashboards, datasources, ... too
158
+ py-tbparse template make sales.twbx # see Templates above
159
+ py-tbparse template apply sales.template.twbx --data q3.csv --write
108
160
  ```
109
161
 
110
- Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`.
162
+ Tables: `overview`, `datasources`, `parameters`, `fields`, `raw-fields`, `calculated-fields`, `joins`, `relations`, `relationships`, `inferred-relationships`, `dashboards`, `dashboard-sheets`, `custom-sql`, `initial-sql`, `published-refs`, `field-usage`, `field-renames`, `report-renames`.
111
163
 
112
- `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
164
+ `--format` takes `table` (default), `csv` or `json`. `graph` always prints Graphviz text. `rename` takes `--reference`, `--datasource`, `--write-workbook [PATH]`, `--style`, `--cutoff`, `--only-changed`, `--all`, `--kinds`, `--apply MAPPING.csv`, `--missing`, `--format` and `--output`. `diff` and `batch` accept the same table names except `graph`, `validate` and `tables`.
113
165
 
114
166
  ## GUI
115
167
 
@@ -136,6 +188,8 @@ pytest
136
188
 
137
189
  The sample workbooks in `tests/fixtures/` come from the R package. `tests/fixtures/public/` holds real workbooks from Tableau's own [document-api-python](https://github.com/tableau/document-api-python) (MIT), used by the smoke tests.
138
190
 
191
+ `tests/corpus/` lists 200 more real workbooks from public repositories with MIT, Apache-2.0, ISC or CC0 licences, for integration tests and as examples (manifest, licence texts and where each file came from are in its README). The files themselves are not in git (about 26 MB): run `python scripts/fetch_corpus.py` to download them, checked against the manifest. `tests/test_corpus.py` then runs every feature over all of them; it skips when they are not fetched.
192
+
139
193
  The GUI tests run the page in headless Chromium through Playwright and fail on any JavaScript error. They skip if the browser isn't installed. To run them:
140
194
 
141
195
  ```bash
@@ -21,8 +21,20 @@ from .rename import (
21
21
  load_rename_mapping,
22
22
  normalize_name,
23
23
  suggest_field_renames,
24
+ suggest_renames,
24
25
  )
25
26
  from .sql import extract_custom_sql, extract_initial_sql
27
+ from .templates import (
28
+ Template,
29
+ TemplateError,
30
+ apply_template,
31
+ load_mapping,
32
+ load_template,
33
+ make_template,
34
+ read_data,
35
+ suggest_mapping,
36
+ )
37
+ from .usage import field_usage
26
38
  from .validators import validate_relationships
27
39
  from ._xml import extract_twb_from_twbx, twbx_extract_files, twbx_list
28
40
 
@@ -56,6 +68,16 @@ __all__ = [
56
68
  "load_rename_mapping",
57
69
  "normalize_name",
58
70
  "suggest_field_renames",
71
+ "suggest_renames",
72
+ "field_usage",
73
+ "Template",
74
+ "TemplateError",
75
+ "make_template",
76
+ "load_template",
77
+ "read_data",
78
+ "suggest_mapping",
79
+ "load_mapping",
80
+ "apply_template",
59
81
  ]
60
82
 
61
83
  try:
@@ -39,12 +39,28 @@ def _calculated_fields(p: TwbParser, include_parameters: bool = False, **_kw) ->
39
39
 
40
40
 
41
41
  def _field_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
42
- datasource=None, **_kw) -> pd.DataFrame:
42
+ datasource=None, kinds=None, **_kw) -> pd.DataFrame:
43
+ if kinds: # the GUI's "everything in the report" switch
44
+ return p.get_renames(
45
+ reference=reference, style=style, only_changed=only_changed, datasource=datasource, kinds=kinds
46
+ )
43
47
  return p.get_field_renames(
44
48
  reference=reference, style=style, only_changed=only_changed, datasource=datasource
45
49
  )
46
50
 
47
51
 
52
+ def _report_renames(p: TwbParser, style: str = "title", reference=None, only_changed: bool = False,
53
+ datasource=None, **_kw) -> pd.DataFrame:
54
+ return p.get_renames(reference=reference, style=style, only_changed=only_changed, datasource=datasource)
55
+
56
+
57
+ def _field_usage(p: TwbParser, **_kw) -> pd.DataFrame:
58
+ df = p.get_field_usage()
59
+ for col in ("sheets", "dashboards", "calculations"):
60
+ df[col] = df[col].map("; ".join)
61
+ return df
62
+
63
+
48
64
  def _joins(p: TwbParser, **_kw) -> pd.DataFrame:
49
65
  return p.get_joins()
50
66
 
@@ -91,6 +107,8 @@ TABLE_SPECS: dict[str, Callable[..., pd.DataFrame]] = {
91
107
  "raw-fields": _raw_fields,
92
108
  "calculated-fields": _calculated_fields,
93
109
  "field-renames": _field_renames,
110
+ "report-renames": _report_renames,
111
+ "field-usage": _field_usage,
94
112
  "joins": _joins,
95
113
  "relations": _relations,
96
114
  "relationships": _relationships,
@@ -10,20 +10,33 @@ from __future__ import annotations
10
10
  import argparse
11
11
  import json
12
12
  import sys
13
+ import zipfile
13
14
  from pathlib import Path
14
15
 
15
16
  import pandas as pd
17
+ from lxml import etree
16
18
 
17
19
  from ._tables import TABLE_NAMES, TABLE_SPECS
18
20
  from .batch import scan_folder
19
21
  from .diff import diff_workbooks
20
22
  from .parser import TwbParser
23
+ from .templates import (
24
+ apply_template,
25
+ broken_sheets,
26
+ load_mapping,
27
+ load_template,
28
+ make_template,
29
+ read_data,
30
+ suggest_mapping,
31
+ )
21
32
  from .rename import (
22
33
  STYLES,
23
34
  apply_field_renames,
24
35
  compare_field_schemas,
36
+ KINDS,
25
37
  load_rename_mapping,
26
38
  suggest_field_renames,
39
+ suggest_renames,
27
40
  )
28
41
 
29
42
 
@@ -165,6 +178,11 @@ def build_rename_arg_parser() -> argparse.ArgumentParser:
165
178
  "--cutoff", type=float, default=0.85,
166
179
  help="0..1 similarity needed to match a reference name approximately (default: 0.85)",
167
180
  )
181
+ ap.add_argument(
182
+ "--kinds", metavar="KIND[,KIND...]",
183
+ help="rename more than fields: any of " + ", ".join(KINDS) + ", or 'all' (default: field)",
184
+ )
185
+ ap.add_argument("--all", action="store_true", help="shorthand for --kinds all")
168
186
  ap.add_argument("--only-changed", action="store_true", help="hide fields that need no rename")
169
187
  ap.add_argument("--datasource", help="only this datasource (its internal name), e.g. the newly added one")
170
188
  ap.add_argument(
@@ -227,8 +245,13 @@ def _run_rename(argv: list[str]) -> int:
227
245
  return 0
228
246
 
229
247
  kwargs = dict(reference=ref, style=args.style, fuzzy_cutoff=args.cutoff, datasource=args.datasource)
248
+ kinds = None
249
+ if args.all or args.kinds:
250
+ kinds = KINDS if args.all or args.kinds.strip() == "all" else tuple(
251
+ k.strip() for k in args.kinds.split(",") if k.strip()
252
+ )
230
253
  try:
231
- everything = suggest_field_renames(wb, **kwargs)
254
+ everything = suggest_renames(wb, kinds=kinds, **kwargs) if kinds else suggest_field_renames(wb, **kwargs)
232
255
  except ValueError as e: # e.g. --cutoff outside 0..1
233
256
  print(f"error: {e}", file=sys.stderr)
234
257
  return 1
@@ -246,6 +269,104 @@ def _run_rename(argv: list[str]) -> int:
246
269
  return 0
247
270
 
248
271
 
272
+ def build_template_arg_parser() -> argparse.ArgumentParser:
273
+ ap = argparse.ArgumentParser(
274
+ prog="py-tbparse template",
275
+ description="Make a reusable template from a workbook, or apply one to new data.",
276
+ )
277
+ sub = ap.add_subparsers(dest="action", required=True)
278
+
279
+ mk = sub.add_parser("make", help="save WORKBOOK as a template (.twbx with a template.json manifest)")
280
+ mk.add_argument("workbook")
281
+ mk.add_argument("--output", "-o", help="template file (default: <name>.template.twbx beside the workbook)")
282
+ mk.add_argument("--name", help="template name (default: the workbook's)")
283
+ mk.add_argument("--description", help="what the template is for")
284
+ mk.add_argument("--keep-data", action="store_true", help="keep extracts and packaged data as sample data")
285
+
286
+ sh = sub.add_parser("show", help="list the fields and parameters a template needs")
287
+ sh.add_argument("template")
288
+ sh.add_argument("--format", "-f", choices=["table", "csv", "json"], default="table")
289
+
290
+ ap_ = sub.add_parser(
291
+ "apply", help="map a template's fields to new data; with --write, make the workbook",
292
+ description="Without --write this only prints the suggested mapping and what would break.",
293
+ )
294
+ ap_.add_argument("template")
295
+ ap_.add_argument("--data", "-d", required=True, help="the new data: a .csv, or a .twb/.twbx/.tds connected to it")
296
+ ap_.add_argument("--datasource", help="which template datasource to fill (when it has several)")
297
+ ap_.add_argument("--data-datasource", help="which datasource of a --data workbook to use")
298
+ ap_.add_argument("--mapping", "-m", help="use this edited mapping CSV instead of the suggestion")
299
+ ap_.add_argument("--mapping-out", help="save the mapping as CSV, to edit and pass back with --mapping")
300
+ ap_.add_argument("--param", "-p", action="append", default=[], metavar="NAME=VALUE",
301
+ help="set a parameter (repeatable), e.g. -p 'Top N=10'")
302
+ ap_.add_argument("--cutoff", type=float, default=0.85, help="0..1 similarity for approximate matches")
303
+ ap_.add_argument("--allow-missing", action="store_true",
304
+ help="write even if required fields have no column (their sheets will break)")
305
+ ap_.add_argument(
306
+ "--write", "-w", nargs="?", const="", metavar="PATH",
307
+ help="make the workbook (default PATH: <template>_<data>.twbx beside the template; never overwrites)",
308
+ )
309
+ ap_.add_argument("--format", "-f", choices=["table", "csv", "json"], default="table")
310
+ return ap
311
+
312
+
313
+ def _run_template(argv: list[str]) -> int:
314
+ ap = build_template_arg_parser()
315
+ args = ap.parse_args(argv)
316
+ try:
317
+ if args.action == "make":
318
+ out = make_template(args.workbook, output_path=args.output, name=args.name,
319
+ description=args.description, keep_data=args.keep_data)
320
+ t = load_template(out)
321
+ req = int(t.fields()["required"].sum())
322
+ print(f"wrote {out} ({req} required field(s), {len(t.parameters())} parameter(s))", file=sys.stderr)
323
+ return 0
324
+
325
+ t = load_template(args.template)
326
+ if args.action == "show":
327
+ if t.manifest.get("description"):
328
+ print(t.manifest["description"], file=sys.stderr)
329
+ _write(_df_text(t.fields(), args.format), None)
330
+ if args.format == "table" and len(t.parameters()):
331
+ print("\nParameters:")
332
+ _write(_df_text(t.parameters(), args.format), None)
333
+ return 0
334
+
335
+ params = {}
336
+ for item in args.param:
337
+ if "=" not in item:
338
+ ap.error(f"--param expects NAME=VALUE, got {item!r}")
339
+ k, v = item.split("=", 1)
340
+ params[k.strip()] = v
341
+ data = read_data(args.data, datasource=args.data_datasource)
342
+ if args.mapping:
343
+ mapping = load_mapping(args.mapping)
344
+ else:
345
+ mapping = suggest_mapping(t, data, datasource=args.datasource, fuzzy_cutoff=args.cutoff)
346
+ _write(_df_text(mapping, args.format), None)
347
+ if args.mapping_out:
348
+ _write(mapping.to_csv(index=False), args.mapping_out)
349
+ broken = broken_sheets(t, mapping, datasource=args.datasource)
350
+ for r in broken.to_dict("records"):
351
+ name = r["caption"] or r["field"].strip("[]")
352
+ print(f"missing: {name} (breaks: {r['sheets'] or 'no sheet'})", file=sys.stderr)
353
+ if args.write is None:
354
+ if not broken.empty:
355
+ print("required fields are unmapped; edit the mapping (--mapping-out / --mapping) "
356
+ "or pass --allow-missing", file=sys.stderr)
357
+ return 0
358
+ report: dict = {}
359
+ out = apply_template(t, data, mapping=mapping, params=params, output_path=args.write or None,
360
+ datasource=args.datasource, allow_missing=args.allow_missing, report=report)
361
+ print(f"wrote {out} ({report['mapped']} field(s) mapped, {report['missing']} missing, "
362
+ f"{report['parameters']} parameter(s) set)", file=sys.stderr)
363
+ return 0
364
+ except (FileNotFoundError, FileExistsError, ValueError, OSError, zipfile.BadZipFile,
365
+ json.JSONDecodeError, etree.XMLSyntaxError, pd.errors.ParserError) as e:
366
+ print(f"error: {e}", file=sys.stderr)
367
+ return 1
368
+
369
+
249
370
  def _is_reserved_subcommand(argv: list[str], name: str) -> bool:
250
371
  """True if `argv` invokes the `name` subcommand -- but don't let that
251
372
  shadow an actual workbook that happens to be named exactly "diff" or
@@ -261,6 +382,8 @@ def main(argv: list[str] | None = None) -> int:
261
382
  return _run_batch(raw_argv[1:])
262
383
  if _is_reserved_subcommand(raw_argv, "rename"):
263
384
  return _run_rename(raw_argv[1:])
385
+ if _is_reserved_subcommand(raw_argv, "template"):
386
+ return _run_template(raw_argv[1:])
264
387
 
265
388
  args = build_arg_parser().parse_args(argv)
266
389
 
@@ -208,6 +208,21 @@ class TwbParser:
208
208
 
209
209
  return suggest_field_renames(self, reference=reference, **kwargs)
210
210
 
211
+ def get_field_usage(self) -> pd.DataFrame:
212
+ """Which sheets, dashboards and calculations use each field; see
213
+ `usage.field_usage`."""
214
+ from .usage import field_usage
215
+
216
+ return field_usage(self)
217
+
218
+ def get_renames(self, reference=None, **kwargs) -> pd.DataFrame:
219
+ """Suggested clean names for everything in the report (fields,
220
+ parameters, worksheets, dashboards, datasources, folders,
221
+ hierarchies); see `rename.suggest_renames`."""
222
+ from .rename import suggest_renames
223
+
224
+ return suggest_renames(self, reference=reference, **kwargs)
225
+
211
226
  def write_renamed_workbook(self, output_path=None, renames=None, overwrite=False, **kwargs) -> str:
212
227
  """Save a copy of the workbook with clean field names; returns its
213
228
  path. See `rename.apply_field_renames`."""