normalize-tabular-data 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,24 @@
1
+ Copyright (c) 2026, Members of Service Employees International Union (SEIU)
2
+ All rights reserved.
3
+
4
+ Redistribution and use in source and binary forms, with or without
5
+ modification, are permitted provided that the following conditions are met:
6
+
7
+ 1. Redistributions of source code must retain the above copyright notice,
8
+ this list of conditions and the following disclaimer.
9
+
10
+ 2. Redistributions in binary form must reproduce the above copyright notice,
11
+ this list of conditions and the following disclaimer in the documentation
12
+ and/or other materials provided with the distribution.
13
+
14
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
15
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
16
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
17
+ ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
18
+ LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
19
+ CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
20
+ SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
21
+ INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
22
+ CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
23
+ ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
24
+ POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,164 @@
1
+ Metadata-Version: 2.4
2
+ Name: normalize-tabular-data
3
+ Version: 0.1.0
4
+ Summary: TUI for normalizing tabular data with polars
5
+ Author: David Mertz, Ph.D.
6
+ Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
7
+ License-Expression: BSD-2-Clause
8
+ License-File: LICENSE
9
+ Requires-Dist: gnosis-date-parser>=1.0.4
10
+ Requires-Dist: textual>=8.2.8
11
+ Requires-Dist: polars>=1.44,<2
12
+ Requires-Dist: fastexcel>=0.10
13
+ Requires-Dist: xlsxwriter>=0.9
14
+ Requires-Python: >=3.14
15
+ Description-Content-Type: text/markdown
16
+
17
+ # normalize-tabular-data
18
+
19
+ A terminal UI (Textual) for interactively normalizing tabular data, backed by
20
+ [polars](https://pola.rs) for speed and
21
+ [gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
22
+ date parsing at Rust speed.
23
+
24
+ Load a file, see a preview, build up a pipeline of normalization operations
25
+ (normalize messy dates, trim whitespace, rename a column, deduplicate,
26
+ combine/split columns, drop columns), then save the
27
+ cleaned result.
28
+
29
+ ## Running
30
+
31
+ ### Persistent install
32
+
33
+ ```bash
34
+ uv tool install normalize-tabular-data
35
+ # upgrade later with:
36
+ uv tool upgrade normalize-tabular-data
37
+ ```
38
+
39
+ ### Ephemeral run (no install)
40
+
41
+ `uvx` fetches into a throwaway environment each time:
42
+
43
+ ```bash
44
+ uvx normalize-tabular-data
45
+ # pin a specific version:
46
+ uvx normalize-tabular-data==0.1.0
47
+ ```
48
+
49
+ ### Straight from a checkout
50
+
51
+ ```bash
52
+ uv run normalize-tabular-data
53
+ ```
54
+
55
+ ### The TestPyPI sandbox
56
+
57
+ The test build (uploaded via `make testpypi`) is not on PyPI; point uv
58
+ at TestPyPI first and PyPI as the fallback index — the tool itself is
59
+ found on TestPyPI, its dependencies (`textual`, `polars`, …) on PyPI:
60
+
61
+ ```bash
62
+ uvx \
63
+ --index-url https://test.pypi.org/simple/ \
64
+ --extra-index-url https://pypi.org/simple/ \
65
+ normalize-tabular-data==0.1.0
66
+
67
+ # or installed persistently:
68
+ uv tool install \
69
+ --index-url https://test.pypi.org/simple/ \
70
+ --extra-index-url https://pypi.org/simple/ \
71
+ normalize-tabular-data==0.1.0
72
+ ```
73
+
74
+ uv caches resolved packages aggressively: after a TestPyPI re-upload of
75
+ the same version, add `--refresh-package normalize-tabular-data` so it
76
+ re-consults the index rather than reusing the cached artifact.
77
+
78
+ ## Usage
79
+
80
+ Launch with `normalize-tabular-data`. Keys:
81
+
82
+ | Key | Action |
83
+ |-----|-----------------------------|
84
+ | `f` | (F)ile — open a file |
85
+ | `o` | (O)peration — choose an op |
86
+ | `p` | (P)lay script — reapply a saved `.ntd` sequence |
87
+ | `u` | (U)ndo last applied step |
88
+ | `r` | (R)edo |
89
+ | `s` | (S)ave the data in a format |
90
+ | `q` | (Q)uit |
91
+
92
+ ## Operations
93
+
94
+ - **Normalize dates** — parse a messy date column of *any* input format into
95
+ canonical UTC datetimes; unparseable values become null.
96
+ - **Trim whitespace** — strip edges and collapse internal whitespace runs,
97
+ per selected columns.
98
+ - Rename a column — click its header in the preview and type the new name.
99
+ - **Deduplicate rows** — on all or selected columns, keeping first or last.
100
+ - **Combine columns** — concatenate two or more columns with a separator.
101
+ - **Split column** — break one column into `{col}_1..{col}_k`; a blank
102
+ delimiter splits on runs of whitespace.
103
+ - **Remove columns** — drop selected columns entirely.
104
+
105
+ ## Formats
106
+
107
+ Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
108
+ Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
109
+
110
+ Large files are previewed with a random sample of 250 rows; every operation
111
+ still runs on the full data.
112
+
113
+ ## Operation scripts (`.ntd`)
114
+
115
+ Every open file keeps an internal log of the operations performed on it. When
116
+ saving, the dialog offers a **Save sequence of operations?** checkbox
117
+ (unticked by default); with it ticked, a second dialog asks where to write
118
+ the script — suggested `<table>.ntd`, any other extension you type is kept.
119
+
120
+ The script is plain ASCII text, one operation per line, e.g.:
121
+
122
+ ```
123
+ # normalize-tabular-data script
124
+ # source: employees.csv
125
+ # table: /data/employees-normalized.csv
126
+ # saved: 2026-10-05T12:30:11
127
+ trim_collapse(columns=["Dept", "Name"])
128
+ date_normalize(column="Hired Date")
129
+ ```
130
+
131
+ ## Playing a script back
132
+
133
+ Press `p` (available while a file is loaded) to pick a script file: the
134
+ dialog previews the highlighted `.ntd` file (syntax-highlighted, first 40
135
+ lines) before you confirm. Its operations are applied, in order, to the table you have open. If any step
136
+ cannot be performed against the currently loaded file — a column it
137
+ renames, trims or splits is missing, the operation is unknown — playing
138
+ stops with an alert naming the failing step, and every step the script had
139
+ already applied is rolled back, so the table is left exactly as it was.
140
+
141
+ ![TUI preview](docs/screenshot.png)
142
+
143
+ ## Publishing
144
+
145
+ PyPI credentials live in `$HOME/.pypirc` (sections `[pypi]` and
146
+ `[testpypi]`); uploads go through `twine`, which honors that file —
147
+ `uv publish` does not read `.pypirc`, so it is not used here.
148
+
149
+ ```bash
150
+ make dist # build sdist+wheel into dist/ and run `twine check`
151
+ make testpypi # upload to the TestPyPI sandbox (login section [testpypi])
152
+ make pypi # upload to PyPI (login section [pypi])
153
+ twine upload # equivalent of `make pypi`, minus a fresh build
154
+ ```
155
+
156
+ The targets fail fast if the package metadata does not pass `twine check`,
157
+ and `twine` runs with `--non-interactive` so a missing or wrong
158
+ credential aborts the upload instead of prompting mid-run.
159
+
160
+ ## License
161
+
162
+ BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
163
+
164
+ See [LICENSE](LICENSE).
@@ -0,0 +1,148 @@
1
+ # normalize-tabular-data
2
+
3
+ A terminal UI (Textual) for interactively normalizing tabular data, backed by
4
+ [polars](https://pola.rs) for speed and
5
+ [gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
6
+ date parsing at Rust speed.
7
+
8
+ Load a file, see a preview, build up a pipeline of normalization operations
9
+ (normalize messy dates, trim whitespace, rename a column, deduplicate,
10
+ combine/split columns, drop columns), then save the
11
+ cleaned result.
12
+
13
+ ## Running
14
+
15
+ ### Persistent install
16
+
17
+ ```bash
18
+ uv tool install normalize-tabular-data
19
+ # upgrade later with:
20
+ uv tool upgrade normalize-tabular-data
21
+ ```
22
+
23
+ ### Ephemeral run (no install)
24
+
25
+ `uvx` fetches into a throwaway environment each time:
26
+
27
+ ```bash
28
+ uvx normalize-tabular-data
29
+ # pin a specific version:
30
+ uvx normalize-tabular-data==0.1.0
31
+ ```
32
+
33
+ ### Straight from a checkout
34
+
35
+ ```bash
36
+ uv run normalize-tabular-data
37
+ ```
38
+
39
+ ### The TestPyPI sandbox
40
+
41
+ The test build (uploaded via `make testpypi`) is not on PyPI; point uv
42
+ at TestPyPI first and PyPI as the fallback index — the tool itself is
43
+ found on TestPyPI, its dependencies (`textual`, `polars`, …) on PyPI:
44
+
45
+ ```bash
46
+ uvx \
47
+ --index-url https://test.pypi.org/simple/ \
48
+ --extra-index-url https://pypi.org/simple/ \
49
+ normalize-tabular-data==0.1.0
50
+
51
+ # or installed persistently:
52
+ uv tool install \
53
+ --index-url https://test.pypi.org/simple/ \
54
+ --extra-index-url https://pypi.org/simple/ \
55
+ normalize-tabular-data==0.1.0
56
+ ```
57
+
58
+ uv caches resolved packages aggressively: after a TestPyPI re-upload of
59
+ the same version, add `--refresh-package normalize-tabular-data` so it
60
+ re-consults the index rather than reusing the cached artifact.
61
+
62
+ ## Usage
63
+
64
+ Launch with `normalize-tabular-data`. Keys:
65
+
66
+ | Key | Action |
67
+ |-----|-----------------------------|
68
+ | `f` | (F)ile — open a file |
69
+ | `o` | (O)peration — choose an op |
70
+ | `p` | (P)lay script — reapply a saved `.ntd` sequence |
71
+ | `u` | (U)ndo last applied step |
72
+ | `r` | (R)edo |
73
+ | `s` | (S)ave the data in a format |
74
+ | `q` | (Q)uit |
75
+
76
+ ## Operations
77
+
78
+ - **Normalize dates** — parse a messy date column of *any* input format into
79
+ canonical UTC datetimes; unparseable values become null.
80
+ - **Trim whitespace** — strip edges and collapse internal whitespace runs,
81
+ per selected columns.
82
+ - Rename a column — click its header in the preview and type the new name.
83
+ - **Deduplicate rows** — on all or selected columns, keeping first or last.
84
+ - **Combine columns** — concatenate two or more columns with a separator.
85
+ - **Split column** — break one column into `{col}_1..{col}_k`; a blank
86
+ delimiter splits on runs of whitespace.
87
+ - **Remove columns** — drop selected columns entirely.
88
+
89
+ ## Formats
90
+
91
+ Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
92
+ Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
93
+
94
+ Large files are previewed with a random sample of 250 rows; every operation
95
+ still runs on the full data.
96
+
97
+ ## Operation scripts (`.ntd`)
98
+
99
+ Every open file keeps an internal log of the operations performed on it. When
100
+ saving, the dialog offers a **Save sequence of operations?** checkbox
101
+ (unticked by default); with it ticked, a second dialog asks where to write
102
+ the script — suggested `<table>.ntd`, any other extension you type is kept.
103
+
104
+ The script is plain ASCII text, one operation per line, e.g.:
105
+
106
+ ```
107
+ # normalize-tabular-data script
108
+ # source: employees.csv
109
+ # table: /data/employees-normalized.csv
110
+ # saved: 2026-10-05T12:30:11
111
+ trim_collapse(columns=["Dept", "Name"])
112
+ date_normalize(column="Hired Date")
113
+ ```
114
+
115
+ ## Playing a script back
116
+
117
+ Press `p` (available while a file is loaded) to pick a script file: the
118
+ dialog previews the highlighted `.ntd` file (syntax-highlighted, first 40
119
+ lines) before you confirm. Its operations are applied, in order, to the table you have open. If any step
120
+ cannot be performed against the currently loaded file — a column it
121
+ renames, trims or splits is missing, the operation is unknown — playing
122
+ stops with an alert naming the failing step, and every step the script had
123
+ already applied is rolled back, so the table is left exactly as it was.
124
+
125
+ ![TUI preview](docs/screenshot.png)
126
+
127
+ ## Publishing
128
+
129
+ PyPI credentials live in `$HOME/.pypirc` (sections `[pypi]` and
130
+ `[testpypi]`); uploads go through `twine`, which honors that file —
131
+ `uv publish` does not read `.pypirc`, so it is not used here.
132
+
133
+ ```bash
134
+ make dist # build sdist+wheel into dist/ and run `twine check`
135
+ make testpypi # upload to the TestPyPI sandbox (login section [testpypi])
136
+ make pypi # upload to PyPI (login section [pypi])
137
+ twine upload # equivalent of `make pypi`, minus a fresh build
138
+ ```
139
+
140
+ The targets fail fast if the package metadata does not pass `twine check`,
141
+ and `twine` runs with `--non-interactive` so a missing or wrong
142
+ credential aborts the upload instead of prompting mid-run.
143
+
144
+ ## License
145
+
146
+ BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
147
+
148
+ See [LICENSE](LICENSE).
@@ -0,0 +1,34 @@
1
+ [project]
2
+ name = "normalize-tabular-data"
3
+ version = "0.1.0"
4
+ description = "TUI for normalizing tabular data with polars"
5
+ readme = "README.md"
6
+ license = "BSD-2-Clause"
7
+ license-files = ["LICENSE"]
8
+ requires-python = ">=3.14"
9
+ dependencies = [
10
+ "gnosis-date-parser>=1.0.4",
11
+ "textual>=8.2.8",
12
+ "polars>=1.44,<2",
13
+ "fastexcel>=0.10",
14
+ "xlsxwriter>=0.9",
15
+ ]
16
+
17
+ [[project.authors]]
18
+ name = "David Mertz, Ph.D."
19
+ email = "mertz@gnosis.cx"
20
+
21
+ [project.scripts]
22
+ normalize-tabular-data = "normalize_tabular_data:main"
23
+
24
+ [dependency-groups]
25
+ dev = [
26
+ "pytest>=8",
27
+ "pytest-asyncio>=0.25",
28
+ "ruff>=0.16.10",
29
+ "twine>=7.0.0",
30
+ ]
31
+
32
+ [build-system]
33
+ requires = ["uv_build>=0.12.3,<0.13.0"]
34
+ build-backend = "uv_build"
@@ -0,0 +1,33 @@
1
+ [project]
2
+ name = "normalize-tabular-data"
3
+ version = "0.1.0"
4
+ description = "TUI for normalizing tabular data with polars"
5
+ readme = "README.md"
6
+ license = "BSD-2-Clause"
7
+ license-files = ["LICENSE"]
8
+ authors = [
9
+ { name = "David Mertz, Ph.D.", email = "mertz@gnosis.cx" }
10
+ ]
11
+ requires-python = ">=3.14"
12
+ dependencies = [
13
+ "gnosis-date-parser>=1.0.4",
14
+ "textual>=8.2.8",
15
+ "polars>=1.44,<2",
16
+ "fastexcel>=0.10",
17
+ "xlsxwriter>=0.9",
18
+ ]
19
+
20
+ [dependency-groups]
21
+ dev = [
22
+ "pytest>=8",
23
+ "pytest-asyncio>=0.25",
24
+ "ruff>=0.16.10",
25
+ "twine>=7.0.0",
26
+ ]
27
+
28
+ [project.scripts]
29
+ normalize-tabular-data = "normalize_tabular_data:main"
30
+
31
+ [build-system]
32
+ requires = ["uv_build>=0.12.3,<0.13.0"]
33
+ build-backend = "uv_build"
@@ -0,0 +1,18 @@
1
+ """normalize-tabular-data: TUI for normalizing tabular data with polars."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+
6
+ def main() -> None:
7
+ try:
8
+ from normalize_tabular_data.app import NormalizeApp
9
+ except ImportError as exc: # helpful message if an optional engine is missing
10
+ import sys
11
+
12
+ print(
13
+ f"normalize-tabular-data is missing a dependency ({exc}).\n"
14
+ "Reinstall with: uv tool install --force --reinstall normalize-tabular-data",
15
+ file=sys.stderr,
16
+ )
17
+ raise SystemExit(1)
18
+ NormalizeApp().run()
@@ -0,0 +1,4 @@
1
+ from normalize_tabular_data import main
2
+
3
+ if __name__ == "__main__":
4
+ main()