normalize-tabular-data 0.1.3__tar.gz → 0.1.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/PKG-INFO +52 -69
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/README.md +51 -68
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/pyproject.toml +1 -1
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/pyproject.toml.orig +1 -1
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/__init__.py +4 -3
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/io.py +1 -1
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/ops.py +1 -1
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/LICENSE +0 -0
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/__main__.py +0 -0
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/app.py +0 -0
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/screens.py +0 -0
- {normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/widgets.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: normalize-tabular-data
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.5
|
|
4
4
|
Summary: TUI for normalizing tabular data with Polars
|
|
5
5
|
Author: David Mertz, Ph.D.
|
|
6
6
|
Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
|
|
@@ -15,55 +15,79 @@ Requires-Dist: platformdirs>=3.0
|
|
|
15
15
|
Requires-Python: >=3.14
|
|
16
16
|
Description-Content-Type: text/markdown
|
|
17
17
|
|
|
18
|
-
# normalize-tabular-data
|
|
18
|
+
# normalize-tabular-data (NTD)
|
|
19
19
|
|
|
20
20
|
A terminal UI (Textual) for interactively normalizing tabular data, backed by
|
|
21
21
|
[polars](https://pola.rs) for speed and
|
|
22
22
|
[gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
|
|
23
|
-
date parsing at Rust speed.
|
|
23
|
+
date parsing at Rust speed. NTD also supports purely command-line operations
|
|
24
|
+
for use in scripting.
|
|
25
|
+
|
|
26
|
+
**Quick start**: `uvx normalize-tabular-data`. No installation required if
|
|
27
|
+
you have `uv`.
|
|
24
28
|
|
|
25
29
|
Load a file, see a preview, build up a pipeline of normalization operations
|
|
26
30
|
(normalize messy dates, trim whitespace, rename a column, deduplicate,
|
|
27
|
-
combine/split columns, drop columns), then save the
|
|
28
|
-
cleaned result.
|
|
31
|
+
combine/split columns, drop columns), then save the cleaned result.
|
|
29
32
|
|
|
30
|
-
|
|
33
|
+
## Formats
|
|
31
34
|
|
|
32
|
-
|
|
35
|
+
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
36
|
+
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
33
37
|
|
|
34
|
-
|
|
38
|
+
Large files are previewed with a random sample of 250 rows; every operation
|
|
39
|
+
still runs on the full data.
|
|
35
40
|
|
|
36
|
-
|
|
37
|
-
uv tool install normalize-tabular-data
|
|
38
|
-
# upgrade later with:
|
|
39
|
-
uv tool upgrade normalize-tabular-data
|
|
40
|
-
```
|
|
41
|
+
## Running the Text User Interface
|
|
41
42
|
|
|
42
|
-
|
|
43
|
+

|
|
43
44
|
|
|
44
|
-
|
|
45
|
+
Installation and launch options — persistent install, ephemeral `uvx`
|
|
46
|
+
runs, running from a git checkout, the single-file launchers for
|
|
47
|
+
Linux/macOS/Windows, and installing them as desktop icons — are
|
|
48
|
+
described in
|
|
49
|
+
[docs/running-the-tool.md](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/docs/running-the-tool.md).
|
|
45
50
|
|
|
46
|
-
|
|
47
|
-
uvx normalize-tabular-data
|
|
48
|
-
# pin a specific version:
|
|
49
|
-
uvx normalize-tabular-data==0.1.1
|
|
50
|
-
```
|
|
51
|
+
## Capabilities in the TUI
|
|
51
52
|
|
|
52
|
-
|
|
53
|
+
Launch with `uvx normalize-tabular-data` (or a launcher). Keys:
|
|
53
54
|
|
|
54
|
-
Within the directory of the cloned repository:
|
|
55
55
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
56
|
+
| Key | Action |
|
|
57
|
+
|-----|------------------------------|
|
|
58
|
+
| `f` | (F)ile — open a file |
|
|
59
|
+
| `o` | (O)peration — choose an op |
|
|
60
|
+
| `p` | (P)lay script — apply `.ntd` |
|
|
61
|
+
| `u` | (U)ndo last applied step |
|
|
62
|
+
| `r` | (R)edo |
|
|
63
|
+
| `s` | (S)ave the data in a format |
|
|
64
|
+
| `q` | (Q)uit |
|
|
59
65
|
|
|
60
|
-
##
|
|
66
|
+
## Operations
|
|
67
|
+
|
|
68
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
69
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
70
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
71
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
72
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
73
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
74
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
75
|
+
(`year-month-day`); unparseable values become null.
|
|
76
|
+
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
77
|
+
per selected columns.
|
|
78
|
+
- Rename a column — click its header in the preview and type the new name.
|
|
79
|
+
- **Combine columns** — concatenate two or more columns with a separator.
|
|
80
|
+
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
81
|
+
delimiter splits on runs of whitespace.
|
|
82
|
+
- **Remove columns** — drop selected columns entirely.
|
|
83
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
84
|
+
|
|
85
|
+
## Command line usage
|
|
61
86
|
|
|
62
87
|
An optional `FILE` argument names a table to open at startup, as if you
|
|
63
88
|
had picked it from the in-app dialog.
|
|
64
89
|
|
|
65
90
|
```bash
|
|
66
|
-
normalize-tabular-data data/members.csv # or with uvx/uv run
|
|
67
91
|
uvx normalize-tabular-data data/members.tsv
|
|
68
92
|
```
|
|
69
93
|
|
|
@@ -104,47 +128,6 @@ to stdout itself and the step lines are suppressed, so stdout carries
|
|
|
104
128
|
only the data. After a successful run the confirmation line goes to
|
|
105
129
|
stderr.
|
|
106
130
|
|
|
107
|
-
## Usage
|
|
108
|
-
|
|
109
|
-
Launch with `normalize-tabular-data`. Keys:
|
|
110
|
-
|
|
111
|
-
| Key | Action |
|
|
112
|
-
|-----|------------------------------|
|
|
113
|
-
| `f` | (F)ile — open a file |
|
|
114
|
-
| `o` | (O)peration — choose an op |
|
|
115
|
-
| `p` | (P)lay script — apply `.ntd` |
|
|
116
|
-
| `u` | (U)ndo last applied step |
|
|
117
|
-
| `r` | (R)edo |
|
|
118
|
-
| `s` | (S)ave the data in a format |
|
|
119
|
-
| `q` | (Q)uit |
|
|
120
|
-
|
|
121
|
-
## Operations
|
|
122
|
-
|
|
123
|
-
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
124
|
-
format into canonical UTC datetimes at millisecond resolution;
|
|
125
|
-
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
126
|
-
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
127
|
-
strings; Parquet and Excel keep the real datetime values.
|
|
128
|
-
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
129
|
-
the UTC calendar date: the column becomes date-only ISO-8601
|
|
130
|
-
(`year-month-day`); unparseable values become null.
|
|
131
|
-
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
132
|
-
per selected columns.
|
|
133
|
-
- Rename a column — click its header in the preview and type the new name.
|
|
134
|
-
- **Combine columns** — concatenate two or more columns with a separator.
|
|
135
|
-
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
136
|
-
delimiter splits on runs of whitespace.
|
|
137
|
-
- **Remove columns** — drop selected columns entirely.
|
|
138
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
139
|
-
|
|
140
|
-
## Formats
|
|
141
|
-
|
|
142
|
-
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
143
|
-
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
144
|
-
|
|
145
|
-
Large files are previewed with a random sample of 250 rows; every operation
|
|
146
|
-
still runs on the full data.
|
|
147
|
-
|
|
148
131
|
## Operation scripts (`.ntd`)
|
|
149
132
|
|
|
150
133
|
Every open file keeps an internal log of the operations performed on it. When
|
|
@@ -200,4 +183,4 @@ credential aborts the upload instead of prompting mid-run.
|
|
|
200
183
|
|
|
201
184
|
BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
|
|
202
185
|
|
|
203
|
-
See [LICENSE](LICENSE).
|
|
186
|
+
See [LICENSE](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/LICENSE).
|
|
@@ -1,52 +1,76 @@
|
|
|
1
|
-
# normalize-tabular-data
|
|
1
|
+
# normalize-tabular-data (NTD)
|
|
2
2
|
|
|
3
3
|
A terminal UI (Textual) for interactively normalizing tabular data, backed by
|
|
4
4
|
[polars](https://pola.rs) for speed and
|
|
5
5
|
[gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
|
|
6
|
-
date parsing at Rust speed.
|
|
6
|
+
date parsing at Rust speed. NTD also supports purely command-line operations
|
|
7
|
+
for use in scripting.
|
|
8
|
+
|
|
9
|
+
**Quick start**: `uvx normalize-tabular-data`. No installation required if
|
|
10
|
+
you have `uv`.
|
|
7
11
|
|
|
8
12
|
Load a file, see a preview, build up a pipeline of normalization operations
|
|
9
13
|
(normalize messy dates, trim whitespace, rename a column, deduplicate,
|
|
10
|
-
combine/split columns, drop columns), then save the
|
|
11
|
-
cleaned result.
|
|
14
|
+
combine/split columns, drop columns), then save the cleaned result.
|
|
12
15
|
|
|
13
|
-
|
|
16
|
+
## Formats
|
|
14
17
|
|
|
15
|
-
|
|
18
|
+
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
19
|
+
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
16
20
|
|
|
17
|
-
|
|
21
|
+
Large files are previewed with a random sample of 250 rows; every operation
|
|
22
|
+
still runs on the full data.
|
|
18
23
|
|
|
19
|
-
|
|
20
|
-
uv tool install normalize-tabular-data
|
|
21
|
-
# upgrade later with:
|
|
22
|
-
uv tool upgrade normalize-tabular-data
|
|
23
|
-
```
|
|
24
|
+
## Running the Text User Interface
|
|
24
25
|
|
|
25
|
-
|
|
26
|
+

|
|
26
27
|
|
|
27
|
-
|
|
28
|
+
Installation and launch options — persistent install, ephemeral `uvx`
|
|
29
|
+
runs, running from a git checkout, the single-file launchers for
|
|
30
|
+
Linux/macOS/Windows, and installing them as desktop icons — are
|
|
31
|
+
described in
|
|
32
|
+
[docs/running-the-tool.md](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/docs/running-the-tool.md).
|
|
28
33
|
|
|
29
|
-
|
|
30
|
-
uvx normalize-tabular-data
|
|
31
|
-
# pin a specific version:
|
|
32
|
-
uvx normalize-tabular-data==0.1.1
|
|
33
|
-
```
|
|
34
|
+
## Capabilities in the TUI
|
|
34
35
|
|
|
35
|
-
|
|
36
|
+
Launch with `uvx normalize-tabular-data` (or a launcher). Keys:
|
|
36
37
|
|
|
37
|
-
Within the directory of the cloned repository:
|
|
38
38
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
39
|
+
| Key | Action |
|
|
40
|
+
|-----|------------------------------|
|
|
41
|
+
| `f` | (F)ile — open a file |
|
|
42
|
+
| `o` | (O)peration — choose an op |
|
|
43
|
+
| `p` | (P)lay script — apply `.ntd` |
|
|
44
|
+
| `u` | (U)ndo last applied step |
|
|
45
|
+
| `r` | (R)edo |
|
|
46
|
+
| `s` | (S)ave the data in a format |
|
|
47
|
+
| `q` | (Q)uit |
|
|
42
48
|
|
|
43
|
-
##
|
|
49
|
+
## Operations
|
|
50
|
+
|
|
51
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
52
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
53
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
54
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
55
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
56
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
57
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
58
|
+
(`year-month-day`); unparseable values become null.
|
|
59
|
+
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
60
|
+
per selected columns.
|
|
61
|
+
- Rename a column — click its header in the preview and type the new name.
|
|
62
|
+
- **Combine columns** — concatenate two or more columns with a separator.
|
|
63
|
+
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
64
|
+
delimiter splits on runs of whitespace.
|
|
65
|
+
- **Remove columns** — drop selected columns entirely.
|
|
66
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
67
|
+
|
|
68
|
+
## Command line usage
|
|
44
69
|
|
|
45
70
|
An optional `FILE` argument names a table to open at startup, as if you
|
|
46
71
|
had picked it from the in-app dialog.
|
|
47
72
|
|
|
48
73
|
```bash
|
|
49
|
-
normalize-tabular-data data/members.csv # or with uvx/uv run
|
|
50
74
|
uvx normalize-tabular-data data/members.tsv
|
|
51
75
|
```
|
|
52
76
|
|
|
@@ -87,47 +111,6 @@ to stdout itself and the step lines are suppressed, so stdout carries
|
|
|
87
111
|
only the data. After a successful run the confirmation line goes to
|
|
88
112
|
stderr.
|
|
89
113
|
|
|
90
|
-
## Usage
|
|
91
|
-
|
|
92
|
-
Launch with `normalize-tabular-data`. Keys:
|
|
93
|
-
|
|
94
|
-
| Key | Action |
|
|
95
|
-
|-----|------------------------------|
|
|
96
|
-
| `f` | (F)ile — open a file |
|
|
97
|
-
| `o` | (O)peration — choose an op |
|
|
98
|
-
| `p` | (P)lay script — apply `.ntd` |
|
|
99
|
-
| `u` | (U)ndo last applied step |
|
|
100
|
-
| `r` | (R)edo |
|
|
101
|
-
| `s` | (S)ave the data in a format |
|
|
102
|
-
| `q` | (Q)uit |
|
|
103
|
-
|
|
104
|
-
## Operations
|
|
105
|
-
|
|
106
|
-
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
107
|
-
format into canonical UTC datetimes at millisecond resolution;
|
|
108
|
-
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
109
|
-
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
110
|
-
strings; Parquet and Excel keep the real datetime values.
|
|
111
|
-
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
112
|
-
the UTC calendar date: the column becomes date-only ISO-8601
|
|
113
|
-
(`year-month-day`); unparseable values become null.
|
|
114
|
-
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
115
|
-
per selected columns.
|
|
116
|
-
- Rename a column — click its header in the preview and type the new name.
|
|
117
|
-
- **Combine columns** — concatenate two or more columns with a separator.
|
|
118
|
-
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
119
|
-
delimiter splits on runs of whitespace.
|
|
120
|
-
- **Remove columns** — drop selected columns entirely.
|
|
121
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
122
|
-
|
|
123
|
-
## Formats
|
|
124
|
-
|
|
125
|
-
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
126
|
-
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
127
|
-
|
|
128
|
-
Large files are previewed with a random sample of 250 rows; every operation
|
|
129
|
-
still runs on the full data.
|
|
130
|
-
|
|
131
114
|
## Operation scripts (`.ntd`)
|
|
132
115
|
|
|
133
116
|
Every open file keeps an internal log of the operations performed on it. When
|
|
@@ -183,4 +166,4 @@ credential aborts the upload instead of prompting mid-run.
|
|
|
183
166
|
|
|
184
167
|
BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
|
|
185
168
|
|
|
186
|
-
See [LICENSE](LICENSE).
|
|
169
|
+
See [LICENSE](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/LICENSE).
|
{normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/__init__.py
RENAMED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""normalize-tabular-data: TUI for normalizing tabular data with
|
|
1
|
+
"""normalize-tabular-data: TUI for normalizing tabular data with Polars."""
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
@@ -6,7 +6,7 @@ import argparse
|
|
|
6
6
|
import sys
|
|
7
7
|
from pathlib import Path
|
|
8
8
|
|
|
9
|
-
__version__ = "0.1.
|
|
9
|
+
__version__ = "0.1.5"
|
|
10
10
|
|
|
11
11
|
|
|
12
12
|
def _run_script(path: Path, script_path: Path, output: Path | None) -> int:
|
|
@@ -145,7 +145,8 @@ def main() -> None:
|
|
|
145
145
|
except ImportError as exc: # helpful message if an optional engine is missing
|
|
146
146
|
print(
|
|
147
147
|
f"normalize-tabular-data is missing a dependency ({exc}).\n"
|
|
148
|
-
"Reinstall with:
|
|
148
|
+
"Reinstall with: "
|
|
149
|
+
"uv tool install --force --reinstall normalize-tabular-data",
|
|
149
150
|
file=sys.stderr,
|
|
150
151
|
)
|
|
151
152
|
raise SystemExit(1)
|
|
File without changes
|
{normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/__main__.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/app.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/screens.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.3 → normalize_tabular_data-0.1.5}/src/normalize_tabular_data/widgets.py
RENAMED
|
File without changes
|