normalize-tabular-data 0.1.1__tar.gz → 0.1.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/LICENSE +1 -1
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/PKG-INFO +81 -19
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/README.md +78 -17
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/pyproject.toml +4 -2
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/pyproject.toml.orig +4 -2
- normalize_tabular_data-0.1.3/src/normalize_tabular_data/__init__.py +152 -0
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/app.py +105 -42
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/io.py +30 -12
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/ops.py +87 -20
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/screens.py +9 -8
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/widgets.py +5 -5
- normalize_tabular_data-0.1.1/src/normalize_tabular_data/__init__.py +0 -18
- {normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/__main__.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: normalize-tabular-data
|
|
3
|
-
Version: 0.1.
|
|
4
|
-
Summary: TUI for normalizing tabular data with
|
|
3
|
+
Version: 0.1.3
|
|
4
|
+
Summary: TUI for normalizing tabular data with Polars
|
|
5
5
|
Author: David Mertz, Ph.D.
|
|
6
6
|
Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
|
|
7
7
|
License-Expression: BSD-2-Clause
|
|
@@ -11,6 +11,7 @@ Requires-Dist: textual>=8.2.8
|
|
|
11
11
|
Requires-Dist: polars>=1.44,<2
|
|
12
12
|
Requires-Dist: fastexcel>=0.10
|
|
13
13
|
Requires-Dist: xlsxwriter>=0.9
|
|
14
|
+
Requires-Dist: platformdirs>=3.0
|
|
14
15
|
Requires-Python: >=3.14
|
|
15
16
|
Description-Content-Type: text/markdown
|
|
16
17
|
|
|
@@ -26,6 +27,8 @@ Load a file, see a preview, build up a pipeline of normalization operations
|
|
|
26
27
|
combine/split columns, drop columns), then save the
|
|
27
28
|
cleaned result.
|
|
28
29
|
|
|
30
|
+

|
|
31
|
+
|
|
29
32
|
## Running
|
|
30
33
|
|
|
31
34
|
### Persistent install
|
|
@@ -43,15 +46,64 @@ uv tool upgrade normalize-tabular-data
|
|
|
43
46
|
```bash
|
|
44
47
|
uvx normalize-tabular-data
|
|
45
48
|
# pin a specific version:
|
|
46
|
-
uvx normalize-tabular-data==0.1.
|
|
49
|
+
uvx normalize-tabular-data==0.1.1
|
|
47
50
|
```
|
|
48
51
|
|
|
49
|
-
###
|
|
52
|
+
### From a git checkout
|
|
53
|
+
|
|
54
|
+
Within the directory of the cloned repository:
|
|
50
55
|
|
|
51
56
|
```bash
|
|
52
57
|
uv run normalize-tabular-data
|
|
53
58
|
```
|
|
54
59
|
|
|
60
|
+
## Command line
|
|
61
|
+
|
|
62
|
+
An optional `FILE` argument names a table to open at startup, as if you
|
|
63
|
+
had picked it from the in-app dialog.
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
normalize-tabular-data data/members.csv # or with uvx/uv run
|
|
67
|
+
uvx normalize-tabular-data data/members.tsv
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
If the file is missing or its format cannot be read, the TUI still
|
|
71
|
+
starts — it shows an alert toast and you can open something else.
|
|
72
|
+
|
|
73
|
+
### Open with a script
|
|
74
|
+
|
|
75
|
+
With `--script/-s`, an `.ntd` script is played on `FILE` as soon as it
|
|
76
|
+
loads (same flow as the `p` key, same stop-and-rollback on a step that
|
|
77
|
+
does not fit the file):
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
uvx normalize-tabular-data data/members.csv -s data/members.ntd
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
### Headless run (`-x`, `--run`)
|
|
84
|
+
|
|
85
|
+
`--run/-x` skips the TUI: it opens `FILE`, plays `SCRIPT` step by step,
|
|
86
|
+
saves the result and exits. Each operation prints to stdout as it is
|
|
87
|
+
applied; anything else — including a failing step, named and rolled
|
|
88
|
+
back exactly as the interactive player would — goes to stderr with a
|
|
89
|
+
nonzero exit code.
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
# saves data/members-normalized.csv next to the original
|
|
93
|
+
uvx normalize-tabular-data members.csv -s members.ntd -x
|
|
94
|
+
|
|
95
|
+
# choose the destination (format taken from its extension):
|
|
96
|
+
uvx normalize-tabular-data members.csv -s members.ntd -x -o out/members.parquet
|
|
97
|
+
|
|
98
|
+
# write the CSV to a pipe instead:
|
|
99
|
+
uvx normalize-tabular-data members.csv -s members.ntd -x -o - | gzip > clean.csv.gz
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
With `-o -` (attached `-o-` works too), the transformed table is written
|
|
103
|
+
to stdout itself and the step lines are suppressed, so stdout carries
|
|
104
|
+
only the data. After a successful run the confirmation line goes to
|
|
105
|
+
stderr.
|
|
106
|
+
|
|
55
107
|
## Usage
|
|
56
108
|
|
|
57
109
|
Launch with `normalize-tabular-data`. Keys:
|
|
@@ -68,16 +120,22 @@ Launch with `normalize-tabular-data`. Keys:
|
|
|
68
120
|
|
|
69
121
|
## Operations
|
|
70
122
|
|
|
71
|
-
- **Normalize
|
|
72
|
-
canonical UTC datetimes
|
|
123
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
124
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
125
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
126
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
127
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
128
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
129
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
130
|
+
(`year-month-day`); unparseable values become null.
|
|
73
131
|
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
74
132
|
per selected columns.
|
|
75
133
|
- Rename a column — click its header in the preview and type the new name.
|
|
76
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
77
134
|
- **Combine columns** — concatenate two or more columns with a separator.
|
|
78
135
|
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
79
136
|
delimiter splits on runs of whitespace.
|
|
80
137
|
- **Remove columns** — drop selected columns entirely.
|
|
138
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
81
139
|
|
|
82
140
|
## Formats
|
|
83
141
|
|
|
@@ -92,30 +150,34 @@ still runs on the full data.
|
|
|
92
150
|
Every open file keeps an internal log of the operations performed on it. When
|
|
93
151
|
saving, the dialog offers a **Save sequence of operations?** checkbox
|
|
94
152
|
(unticked by default); with it ticked, a second dialog asks where to write
|
|
95
|
-
the script — suggested `<table>.ntd
|
|
153
|
+
the script — suggested `<table>.ntd`; scripts always use the `.ntd`
|
|
154
|
+
extension, so the name you type is saved with that suffix whatever it
|
|
155
|
+
ends in.
|
|
96
156
|
|
|
97
157
|
The script is plain ASCII text, one operation per line, e.g.:
|
|
98
158
|
|
|
99
159
|
```
|
|
100
160
|
# normalize-tabular-data script
|
|
101
|
-
# source:
|
|
102
|
-
# table: /data/
|
|
161
|
+
# source: worksite.csv
|
|
162
|
+
# table: /data/worksite-normalized.csv
|
|
103
163
|
# saved: 2026-10-05T12:30:11
|
|
104
|
-
|
|
105
|
-
|
|
164
|
+
date_normalize(column="Signed Up")
|
|
165
|
+
trim_collapse(columns=["Full Name", "Worksite", "Job Class"])
|
|
166
|
+
split_column(column="Full Name", delimiter="")
|
|
167
|
+
rename_single(column="Full Name_1", new_name="First Name")
|
|
168
|
+
rename_single(column="Full Name_2", new_name="Last Name")
|
|
106
169
|
```
|
|
107
170
|
|
|
108
171
|
## Playing a script back
|
|
109
172
|
|
|
110
173
|
Press `p` (available while a file is loaded) to pick a script file: the
|
|
111
174
|
dialog previews the highlighted `.ntd` file (syntax-highlighted, first 40
|
|
112
|
-
lines) before you confirm. Its operations are applied, in order, to the table
|
|
113
|
-
cannot be performed against the currently loaded
|
|
114
|
-
renames, trims or splits is missing, the operation is
|
|
115
|
-
stops with an alert naming the failing step, and every step
|
|
116
|
-
already applied is rolled back, so the table is left exactly as
|
|
117
|
-
|
|
118
|
-

|
|
175
|
+
lines) before you confirm. Its operations are applied, in order, to the table
|
|
176
|
+
you have open. If any step cannot be performed against the currently loaded
|
|
177
|
+
file — a column it renames, trims or splits is missing, the operation is
|
|
178
|
+
unknown — playing stops with an alert naming the failing step, and every step
|
|
179
|
+
the script had already applied is rolled back, so the table is left exactly as
|
|
180
|
+
it was.
|
|
119
181
|
|
|
120
182
|
## Publishing
|
|
121
183
|
|
|
@@ -10,6 +10,8 @@ Load a file, see a preview, build up a pipeline of normalization operations
|
|
|
10
10
|
combine/split columns, drop columns), then save the
|
|
11
11
|
cleaned result.
|
|
12
12
|
|
|
13
|
+

|
|
14
|
+
|
|
13
15
|
## Running
|
|
14
16
|
|
|
15
17
|
### Persistent install
|
|
@@ -27,15 +29,64 @@ uv tool upgrade normalize-tabular-data
|
|
|
27
29
|
```bash
|
|
28
30
|
uvx normalize-tabular-data
|
|
29
31
|
# pin a specific version:
|
|
30
|
-
uvx normalize-tabular-data==0.1.
|
|
32
|
+
uvx normalize-tabular-data==0.1.1
|
|
31
33
|
```
|
|
32
34
|
|
|
33
|
-
###
|
|
35
|
+
### From a git checkout
|
|
36
|
+
|
|
37
|
+
Within the directory of the cloned repository:
|
|
34
38
|
|
|
35
39
|
```bash
|
|
36
40
|
uv run normalize-tabular-data
|
|
37
41
|
```
|
|
38
42
|
|
|
43
|
+
## Command line
|
|
44
|
+
|
|
45
|
+
An optional `FILE` argument names a table to open at startup, as if you
|
|
46
|
+
had picked it from the in-app dialog.
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
normalize-tabular-data data/members.csv # or with uvx/uv run
|
|
50
|
+
uvx normalize-tabular-data data/members.tsv
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
If the file is missing or its format cannot be read, the TUI still
|
|
54
|
+
starts — it shows an alert toast and you can open something else.
|
|
55
|
+
|
|
56
|
+
### Open with a script
|
|
57
|
+
|
|
58
|
+
With `--script/-s`, an `.ntd` script is played on `FILE` as soon as it
|
|
59
|
+
loads (same flow as the `p` key, same stop-and-rollback on a step that
|
|
60
|
+
does not fit the file):
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
uvx normalize-tabular-data data/members.csv -s data/members.ntd
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
### Headless run (`-x`, `--run`)
|
|
67
|
+
|
|
68
|
+
`--run/-x` skips the TUI: it opens `FILE`, plays `SCRIPT` step by step,
|
|
69
|
+
saves the result and exits. Each operation prints to stdout as it is
|
|
70
|
+
applied; anything else — including a failing step, named and rolled
|
|
71
|
+
back exactly as the interactive player would — goes to stderr with a
|
|
72
|
+
nonzero exit code.
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
# saves data/members-normalized.csv next to the original
|
|
76
|
+
uvx normalize-tabular-data members.csv -s members.ntd -x
|
|
77
|
+
|
|
78
|
+
# choose the destination (format taken from its extension):
|
|
79
|
+
uvx normalize-tabular-data members.csv -s members.ntd -x -o out/members.parquet
|
|
80
|
+
|
|
81
|
+
# write the CSV to a pipe instead:
|
|
82
|
+
uvx normalize-tabular-data members.csv -s members.ntd -x -o - | gzip > clean.csv.gz
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
With `-o -` (attached `-o-` works too), the transformed table is written
|
|
86
|
+
to stdout itself and the step lines are suppressed, so stdout carries
|
|
87
|
+
only the data. After a successful run the confirmation line goes to
|
|
88
|
+
stderr.
|
|
89
|
+
|
|
39
90
|
## Usage
|
|
40
91
|
|
|
41
92
|
Launch with `normalize-tabular-data`. Keys:
|
|
@@ -52,16 +103,22 @@ Launch with `normalize-tabular-data`. Keys:
|
|
|
52
103
|
|
|
53
104
|
## Operations
|
|
54
105
|
|
|
55
|
-
- **Normalize
|
|
56
|
-
canonical UTC datetimes
|
|
106
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
107
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
108
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
109
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
110
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
111
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
112
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
113
|
+
(`year-month-day`); unparseable values become null.
|
|
57
114
|
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
58
115
|
per selected columns.
|
|
59
116
|
- Rename a column — click its header in the preview and type the new name.
|
|
60
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
61
117
|
- **Combine columns** — concatenate two or more columns with a separator.
|
|
62
118
|
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
63
119
|
delimiter splits on runs of whitespace.
|
|
64
120
|
- **Remove columns** — drop selected columns entirely.
|
|
121
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
65
122
|
|
|
66
123
|
## Formats
|
|
67
124
|
|
|
@@ -76,30 +133,34 @@ still runs on the full data.
|
|
|
76
133
|
Every open file keeps an internal log of the operations performed on it. When
|
|
77
134
|
saving, the dialog offers a **Save sequence of operations?** checkbox
|
|
78
135
|
(unticked by default); with it ticked, a second dialog asks where to write
|
|
79
|
-
the script — suggested `<table>.ntd
|
|
136
|
+
the script — suggested `<table>.ntd`; scripts always use the `.ntd`
|
|
137
|
+
extension, so the name you type is saved with that suffix whatever it
|
|
138
|
+
ends in.
|
|
80
139
|
|
|
81
140
|
The script is plain ASCII text, one operation per line, e.g.:
|
|
82
141
|
|
|
83
142
|
```
|
|
84
143
|
# normalize-tabular-data script
|
|
85
|
-
# source:
|
|
86
|
-
# table: /data/
|
|
144
|
+
# source: worksite.csv
|
|
145
|
+
# table: /data/worksite-normalized.csv
|
|
87
146
|
# saved: 2026-10-05T12:30:11
|
|
88
|
-
|
|
89
|
-
|
|
147
|
+
date_normalize(column="Signed Up")
|
|
148
|
+
trim_collapse(columns=["Full Name", "Worksite", "Job Class"])
|
|
149
|
+
split_column(column="Full Name", delimiter="")
|
|
150
|
+
rename_single(column="Full Name_1", new_name="First Name")
|
|
151
|
+
rename_single(column="Full Name_2", new_name="Last Name")
|
|
90
152
|
```
|
|
91
153
|
|
|
92
154
|
## Playing a script back
|
|
93
155
|
|
|
94
156
|
Press `p` (available while a file is loaded) to pick a script file: the
|
|
95
157
|
dialog previews the highlighted `.ntd` file (syntax-highlighted, first 40
|
|
96
|
-
lines) before you confirm. Its operations are applied, in order, to the table
|
|
97
|
-
cannot be performed against the currently loaded
|
|
98
|
-
renames, trims or splits is missing, the operation is
|
|
99
|
-
stops with an alert naming the failing step, and every step
|
|
100
|
-
already applied is rolled back, so the table is left exactly as
|
|
101
|
-
|
|
102
|
-

|
|
158
|
+
lines) before you confirm. Its operations are applied, in order, to the table
|
|
159
|
+
you have open. If any step cannot be performed against the currently loaded
|
|
160
|
+
file — a column it renames, trims or splits is missing, the operation is
|
|
161
|
+
unknown — playing stops with an alert naming the failing step, and every step
|
|
162
|
+
the script had already applied is rolled back, so the table is left exactly as
|
|
163
|
+
it was.
|
|
103
164
|
|
|
104
165
|
## Publishing
|
|
105
166
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.1.
|
|
4
|
-
description = "TUI for normalizing tabular data with
|
|
3
|
+
version = "0.1.3"
|
|
4
|
+
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
@@ -12,6 +12,7 @@ dependencies = [
|
|
|
12
12
|
"polars>=1.44,<2",
|
|
13
13
|
"fastexcel>=0.10",
|
|
14
14
|
"xlsxwriter>=0.9",
|
|
15
|
+
"platformdirs>=3.0",
|
|
15
16
|
]
|
|
16
17
|
|
|
17
18
|
[[project.authors]]
|
|
@@ -27,6 +28,7 @@ dev = [
|
|
|
27
28
|
"pytest-asyncio>=0.25",
|
|
28
29
|
"ruff>=0.16.10",
|
|
29
30
|
"twine>=7.0.0",
|
|
31
|
+
"ty>=0.0.84",
|
|
30
32
|
]
|
|
31
33
|
|
|
32
34
|
[build-system]
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.1.
|
|
4
|
-
description = "TUI for normalizing tabular data with
|
|
3
|
+
version = "0.1.3"
|
|
4
|
+
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
@@ -15,6 +15,7 @@ dependencies = [
|
|
|
15
15
|
"polars>=1.44,<2",
|
|
16
16
|
"fastexcel>=0.10",
|
|
17
17
|
"xlsxwriter>=0.9",
|
|
18
|
+
"platformdirs>=3.0",
|
|
18
19
|
]
|
|
19
20
|
|
|
20
21
|
[dependency-groups]
|
|
@@ -23,6 +24,7 @@ dev = [
|
|
|
23
24
|
"pytest-asyncio>=0.25",
|
|
24
25
|
"ruff>=0.16.10",
|
|
25
26
|
"twine>=7.0.0",
|
|
27
|
+
"ty>=0.0.84",
|
|
26
28
|
]
|
|
27
29
|
|
|
28
30
|
[project.scripts]
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""normalize-tabular-data: TUI for normalizing tabular data with polars."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
__version__ = "0.1.3"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _run_script(path: Path, script_path: Path, output: Path | None) -> int:
|
|
13
|
+
"""`--run` mode: play SCRIPT against FILE and save the result, no TUI.
|
|
14
|
+
|
|
15
|
+
Operation steps print to STDOUT as they are applied (suppressed when
|
|
16
|
+
STDOUT itself is the output, so the only thing there is the table);
|
|
17
|
+
every other message, including any step failure, goes to STDERR and
|
|
18
|
+
exits nonzero. A failed script leaves nothing written."""
|
|
19
|
+
from normalize_tabular_data import io
|
|
20
|
+
from normalize_tabular_data.ops import Pipeline, apply_script_steps
|
|
21
|
+
|
|
22
|
+
try:
|
|
23
|
+
steps = io.read_script(script_path)
|
|
24
|
+
except Exception as exc:
|
|
25
|
+
print(f"Could not read {script_path.name}: {exc}", file=sys.stderr)
|
|
26
|
+
return 1
|
|
27
|
+
if not steps:
|
|
28
|
+
print(f"{script_path.name} contains no operations", file=sys.stderr)
|
|
29
|
+
return 1
|
|
30
|
+
try:
|
|
31
|
+
fmt = io.detect_format(path)
|
|
32
|
+
table = io.read_table(path, fmt)
|
|
33
|
+
except Exception as exc:
|
|
34
|
+
print(f"Could not read {path.name}: {exc}", file=sys.stderr)
|
|
35
|
+
return 1
|
|
36
|
+
|
|
37
|
+
pipeline = Pipeline(source=table)
|
|
38
|
+
op_log: list[tuple[str, dict]] = []
|
|
39
|
+
to_stdout = output is not None and str(output) == "-"
|
|
40
|
+
if not to_stdout:
|
|
41
|
+
for key, params in steps:
|
|
42
|
+
print(io.step_line(key, params))
|
|
43
|
+
|
|
44
|
+
def fail(reason: str) -> None:
|
|
45
|
+
print(reason, file=sys.stderr)
|
|
46
|
+
|
|
47
|
+
if not apply_script_steps(steps, pipeline, op_log, fail):
|
|
48
|
+
return 1
|
|
49
|
+
# the transformed frame lives in the pipeline; the `table` variable
|
|
50
|
+
# still holds the raw source
|
|
51
|
+
table = pipeline.current()
|
|
52
|
+
|
|
53
|
+
out_path = (
|
|
54
|
+
None if to_stdout else (output or path.with_stem(path.stem + "-normalized"))
|
|
55
|
+
)
|
|
56
|
+
try:
|
|
57
|
+
if out_path is None:
|
|
58
|
+
if fmt not in ("csv", "tsv", "jsonl"):
|
|
59
|
+
print(
|
|
60
|
+
f"{fmt!r} is not a text format: it cannot be written to "
|
|
61
|
+
"STDOUT. Name an --output file with a csv/tsv/jsonl "
|
|
62
|
+
"extension instead.",
|
|
63
|
+
file=sys.stderr,
|
|
64
|
+
)
|
|
65
|
+
return 1
|
|
66
|
+
if fmt == "csv":
|
|
67
|
+
table.write_csv(sys.stdout)
|
|
68
|
+
elif fmt == "tsv":
|
|
69
|
+
table.write_csv(sys.stdout, separator="\t")
|
|
70
|
+
else:
|
|
71
|
+
table.write_ndjson(sys.stdout)
|
|
72
|
+
else:
|
|
73
|
+
out_fmt = io.detect_format(out_path)
|
|
74
|
+
if out_path.suffix.lower() == ".xls":
|
|
75
|
+
print(
|
|
76
|
+
".xls cannot be written; use --output <name>.xlsx", file=sys.stderr
|
|
77
|
+
)
|
|
78
|
+
return 1
|
|
79
|
+
io.write_table(table, out_path, out_fmt)
|
|
80
|
+
except Exception as exc:
|
|
81
|
+
print(f"Save failed: {exc}", file=sys.stderr)
|
|
82
|
+
return 1
|
|
83
|
+
|
|
84
|
+
count = len(steps)
|
|
85
|
+
landing = "STDOUT" if out_path is None else str(out_path)
|
|
86
|
+
print(
|
|
87
|
+
f"Applied {count} step{'s' if count != 1 else ''} from "
|
|
88
|
+
f"{script_path.name} -> {landing}",
|
|
89
|
+
file=sys.stderr,
|
|
90
|
+
)
|
|
91
|
+
return 0
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def main() -> None:
|
|
95
|
+
parser = argparse.ArgumentParser(
|
|
96
|
+
prog="normalize-tabular-data",
|
|
97
|
+
description="A terminal UI for normalizing tabular data with polars.",
|
|
98
|
+
)
|
|
99
|
+
parser.add_argument(
|
|
100
|
+
"file",
|
|
101
|
+
nargs="?",
|
|
102
|
+
type=Path,
|
|
103
|
+
metavar="FILE",
|
|
104
|
+
help="table file to open at startup (csv, tsv, jsonl, parquet, xlsx)",
|
|
105
|
+
)
|
|
106
|
+
parser.add_argument(
|
|
107
|
+
"-s",
|
|
108
|
+
"--script",
|
|
109
|
+
type=Path,
|
|
110
|
+
metavar="SCRIPT",
|
|
111
|
+
help="an .ntd operation script to play on FILE after opening it",
|
|
112
|
+
)
|
|
113
|
+
parser.add_argument(
|
|
114
|
+
"-o",
|
|
115
|
+
"--output",
|
|
116
|
+
type=Path,
|
|
117
|
+
metavar="PATH",
|
|
118
|
+
help="where --run saves the transformed table: a path whose format "
|
|
119
|
+
"is its extension, or - for STDOUT (default: the FILE stem plus "
|
|
120
|
+
"-normalized, same directory and extension)",
|
|
121
|
+
)
|
|
122
|
+
parser.add_argument(
|
|
123
|
+
"-x",
|
|
124
|
+
"--run",
|
|
125
|
+
action="store_true",
|
|
126
|
+
help="open FILE, play SCRIPT step by step, save the result and "
|
|
127
|
+
"exit (no UI; requires FILE and --script)",
|
|
128
|
+
)
|
|
129
|
+
argv = sys.argv[1:]
|
|
130
|
+
args = parser.parse_args(argv)
|
|
131
|
+
|
|
132
|
+
if args.script is not None and args.file is None:
|
|
133
|
+
parser.error("--script needs a FILE to open")
|
|
134
|
+
if args.output is not None and not args.run:
|
|
135
|
+
parser.error("--output only applies with --run")
|
|
136
|
+
if args.run:
|
|
137
|
+
if args.file is None:
|
|
138
|
+
parser.error("--run needs a FILE to open")
|
|
139
|
+
if args.script is None:
|
|
140
|
+
parser.error("--run needs --script (there would be nothing to play)")
|
|
141
|
+
raise SystemExit(_run_script(args.file, args.script, args.output))
|
|
142
|
+
|
|
143
|
+
try:
|
|
144
|
+
from normalize_tabular_data.app import NormalizeApp
|
|
145
|
+
except ImportError as exc: # helpful message if an optional engine is missing
|
|
146
|
+
print(
|
|
147
|
+
f"normalize-tabular-data is missing a dependency ({exc}).\n"
|
|
148
|
+
"Reinstall with: uv tool install --force --reinstall normalize-tabular-data",
|
|
149
|
+
file=sys.stderr,
|
|
150
|
+
)
|
|
151
|
+
raise SystemExit(1)
|
|
152
|
+
NormalizeApp(initial_file=args.file, initial_script=args.script).run()
|
{normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/app.py
RENAMED
|
@@ -6,6 +6,7 @@ import bisect
|
|
|
6
6
|
import datetime as _dt
|
|
7
7
|
import random
|
|
8
8
|
from pathlib import Path
|
|
9
|
+
from platformdirs import user_config_dir
|
|
9
10
|
from typing import Any, Iterable
|
|
10
11
|
|
|
11
12
|
from textual import on
|
|
@@ -28,6 +29,7 @@ from normalize_tabular_data.ops import (
|
|
|
28
29
|
Operation,
|
|
29
30
|
Pipeline,
|
|
30
31
|
analyze_column,
|
|
32
|
+
apply_script_steps as ops_apply_script_steps,
|
|
31
33
|
)
|
|
32
34
|
from normalize_tabular_data.screens import (
|
|
33
35
|
ModalDialog,
|
|
@@ -42,6 +44,7 @@ from normalize_tabular_data.screens import (
|
|
|
42
44
|
from normalize_tabular_data.widgets import ColumnSidebar, MenuFooter, StepsBar
|
|
43
45
|
|
|
44
46
|
PREVIEW_ROWS = 250
|
|
47
|
+
APP_CONFIG_NAME = "normalize-tabular-data"
|
|
45
48
|
|
|
46
49
|
|
|
47
50
|
class CommandMenu(CommandPalette):
|
|
@@ -134,10 +137,19 @@ class OpChooserModal(ModalDialog):
|
|
|
134
137
|
|
|
135
138
|
def _marked_title(self, op_key: str, title: str, hotkey: str = "") -> str:
|
|
136
139
|
"""Return "Normalize (d)ates"-style markup: the op's hotkey letter
|
|
137
|
-
highlighted where it sits in the title (keeping its case).
|
|
138
|
-
|
|
139
|
-
|
|
140
|
+
highlighted where it sits in the title (keeping its case). A title
|
|
141
|
+
that already parenthesizes its letter ("Normalize date(t)imes")
|
|
142
|
+
highlights the letter between the parens; a plain title matches
|
|
143
|
+
its first occurrence and adds the parens. Ops without a designated
|
|
144
|
+
hotkey claim the first unused letter in the title instead; plain
|
|
145
|
+
title if no letter can be claimed."""
|
|
140
146
|
if hotkey:
|
|
147
|
+
i = title.lower().find(f"({hotkey.lower()})")
|
|
148
|
+
if i >= 0:
|
|
149
|
+
# the letter to claim sits inside existing parentheses:
|
|
150
|
+
# highlight it without adding a second pair
|
|
151
|
+
self._hotkeys[title[i + 1].lower()] = op_key
|
|
152
|
+
return f"{title[: i + 1]}[cyan]{title[i + 1]}[/cyan]{title[i + 2:]}"
|
|
141
153
|
i = title.lower().find(hotkey.lower())
|
|
142
154
|
else:
|
|
143
155
|
i = next(
|
|
@@ -270,8 +282,29 @@ class NormalizeApp(App[None]):
|
|
|
270
282
|
}
|
|
271
283
|
"""
|
|
272
284
|
|
|
273
|
-
def __init__(
|
|
285
|
+
def __init__(
|
|
286
|
+
self,
|
|
287
|
+
initial_file: str | Path | None = None,
|
|
288
|
+
initial_script: str | Path | None = None,
|
|
289
|
+
config_dir: Path | None = None,
|
|
290
|
+
) -> None:
|
|
274
291
|
super().__init__()
|
|
292
|
+
# where the chosen theme (and any future settings) are stored;
|
|
293
|
+
# None means the per-user platform directory
|
|
294
|
+
self._config_dir = config_dir
|
|
295
|
+
# flips to True at on_mount: only theme changes made after are the
|
|
296
|
+
# user's and only those are persisted
|
|
297
|
+
self._app_ready = False
|
|
298
|
+
# filesystem path to open when the app starts (the optional file
|
|
299
|
+
# named on the command line); loaded like a file chosen in-app
|
|
300
|
+
self.initial_file: Path | None = (
|
|
301
|
+
Path(initial_file) if initial_file is not None else None
|
|
302
|
+
)
|
|
303
|
+
# .ntd script played automatically once that file is loaded (the
|
|
304
|
+
# optional --script named on the command line); requires a file
|
|
305
|
+
self.initial_script: Path | None = (
|
|
306
|
+
Path(initial_script) if initial_script is not None else None
|
|
307
|
+
)
|
|
275
308
|
self.path: Path | None = None
|
|
276
309
|
self.pipeline: Pipeline | None = None
|
|
277
310
|
self.analyzed: dict[str, Any] = {}
|
|
@@ -339,8 +372,56 @@ class NormalizeApp(App[None]):
|
|
|
339
372
|
return list(self.pipeline.current().columns)
|
|
340
373
|
|
|
341
374
|
def on_mount(self) -> None:
|
|
375
|
+
# remember this point so theme changes from here on are the
|
|
376
|
+
# user's own choices and get persisted
|
|
377
|
+
self._apply_saved_theme()
|
|
342
378
|
self.push_screen(MainScreen())
|
|
343
379
|
self._refresh_steps()
|
|
380
|
+
# the command line's file, if any, is opened once the main screen
|
|
381
|
+
# is composed; load_path toasts a readable alert instead of
|
|
382
|
+
# crashing if it is unreadable or has an unsupported format
|
|
383
|
+
if self.initial_file is not None:
|
|
384
|
+
self.call_after_refresh(self.load_path, self.initial_file)
|
|
385
|
+
self._app_ready = True
|
|
386
|
+
|
|
387
|
+
# --- theme persistence ---------------------------------------------
|
|
388
|
+
|
|
389
|
+
def watch_theme(self, theme_name: str) -> None:
|
|
390
|
+
"""Persist every theme the user chooses (menu or command), so the
|
|
391
|
+
next launch opens with it again. Skipped while headless, so the
|
|
392
|
+
pilot tests cannot touch the real configuration; skipped before
|
|
393
|
+
mount, where only the reactive's default applies."""
|
|
394
|
+
if self._app_ready and not self.is_headless:
|
|
395
|
+
try:
|
|
396
|
+
self._theme_config_path().parent.mkdir(parents=True, exist_ok=True)
|
|
397
|
+
self._theme_config_path().write_text(
|
|
398
|
+
theme_name + "\n", encoding="utf-8"
|
|
399
|
+
)
|
|
400
|
+
except OSError:
|
|
401
|
+
pass # unwritable config location: run without remembering
|
|
402
|
+
|
|
403
|
+
def _apply_saved_theme(self) -> None:
|
|
404
|
+
"""Restore the theme chosen in a previous session, if it still exists.
|
|
405
|
+
|
|
406
|
+
Skipped in headless runs that use the real configuration location
|
|
407
|
+
(pilot tests): an eyeballing developer's own saved theme should
|
|
408
|
+
not leak into supposedly default-themed tests. A headless run
|
|
409
|
+
given an explicit config_dir is a deliberate fixture and loads it."""
|
|
410
|
+
if self.is_headless and self._config_dir is None:
|
|
411
|
+
return
|
|
412
|
+
try:
|
|
413
|
+
saved = self._theme_config_path().read_text(encoding="utf-8").strip()
|
|
414
|
+
except OSError:
|
|
415
|
+
return
|
|
416
|
+
if not saved:
|
|
417
|
+
return
|
|
418
|
+
try:
|
|
419
|
+
self.theme = saved
|
|
420
|
+
except Exception:
|
|
421
|
+
pass # a theme name that no longer exists: keep the default
|
|
422
|
+
|
|
423
|
+
def _theme_config_path(self) -> Path:
|
|
424
|
+
return (self._config_dir or Path(user_config_dir(APP_CONFIG_NAME))) / "theme"
|
|
344
425
|
|
|
345
426
|
# --- data plumbing ---------------------------------------------------
|
|
346
427
|
|
|
@@ -386,6 +467,11 @@ class NormalizeApp(App[None]):
|
|
|
386
467
|
# pipeline is now set: re-evaluate the footer's enabled keys
|
|
387
468
|
self.screen.refresh_bindings()
|
|
388
469
|
self.notify(f"Loaded {path.name} ({df.height:,} rows x {df.width} columns)")
|
|
470
|
+
# --script named on the command line: play it as soon as its file
|
|
471
|
+
# is on screen (same flow as the P key, with the same rollback)
|
|
472
|
+
if self.initial_script is not None:
|
|
473
|
+
self.play_script_file(self.initial_script)
|
|
474
|
+
self.initial_script = None
|
|
389
475
|
|
|
390
476
|
def analyze_columns(self) -> None:
|
|
391
477
|
if self.pipeline is None:
|
|
@@ -499,50 +585,27 @@ class NormalizeApp(App[None]):
|
|
|
499
585
|
return
|
|
500
586
|
started_applied = len(self.pipeline.applied)
|
|
501
587
|
started_log = len(self.op_log)
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
except Exception as exc:
|
|
518
|
-
self._rollback_script(
|
|
519
|
-
index,
|
|
520
|
-
started_applied,
|
|
521
|
-
started_log,
|
|
522
|
-
f"Step {index} of {len(steps)} ({op.title}) failed: {exc}",
|
|
523
|
-
)
|
|
524
|
-
return
|
|
525
|
-
self.op_log.append((key, dict(params)))
|
|
588
|
+
|
|
589
|
+
def alert(reason: str) -> None:
|
|
590
|
+
# same alert the player has always shown: step description plus
|
|
591
|
+
# how much of the script had to be rolled back
|
|
592
|
+
undid = len(self.pipeline.applied) - started_applied
|
|
593
|
+
detail = (
|
|
594
|
+
f" — rolled back {undid} step{'s' if undid != 1 else ''}"
|
|
595
|
+
if undid
|
|
596
|
+
else ""
|
|
597
|
+
)
|
|
598
|
+
self.refresh_all()
|
|
599
|
+
self.notify(f"{reason}{detail}", severity="error")
|
|
600
|
+
|
|
601
|
+
if not ops_apply_script_steps(steps, self.pipeline, self.op_log, alert):
|
|
602
|
+
return
|
|
526
603
|
self.refresh_all()
|
|
527
604
|
count = len(steps)
|
|
528
605
|
self.notify(
|
|
529
606
|
f"Applied {count} step{'s' if count != 1 else ''} from {script_path.name}"
|
|
530
607
|
)
|
|
531
608
|
|
|
532
|
-
def _rollback_script(
|
|
533
|
-
self, step: int, started_applied: int, started_log: int, reason: str
|
|
534
|
-
) -> None:
|
|
535
|
-
"""Undo everything a partially played script applied, then alert."""
|
|
536
|
-
undid = len(self.pipeline.applied) - started_applied
|
|
537
|
-
del self.pipeline.applied[started_applied:]
|
|
538
|
-
self.pipeline.redo_stack.clear()
|
|
539
|
-
del self.op_log[started_log:]
|
|
540
|
-
self.refresh_all()
|
|
541
|
-
detail = (
|
|
542
|
-
f" — rolled back {undid} step{'s' if undid != 1 else ''}" if undid else ""
|
|
543
|
-
)
|
|
544
|
-
self.notify(f"{reason}{detail}", severity="error")
|
|
545
|
-
|
|
546
609
|
def _apply_now(self, op: Operation, params: dict[str, Any]) -> None:
|
|
547
610
|
try:
|
|
548
611
|
self.pipeline.apply(op, params)
|
{normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/io.py
RENAMED
|
@@ -79,6 +79,24 @@ def excel_sheets(path: Path) -> list[str]:
|
|
|
79
79
|
return list(fastexcel.read_excel(path).sheet_names)
|
|
80
80
|
|
|
81
81
|
|
|
82
|
+
def _second_resolution(df: pl.DataFrame) -> pl.DataFrame:
|
|
83
|
+
"""Datetime columns serialized as canonical UTC second-resolution text.
|
|
84
|
+
|
|
85
|
+
Used only for the text formats (CSV/TSV/JSONL): polars pads datetime
|
|
86
|
+
output with sub-second digits no matter the unit, and the canonical
|
|
87
|
+
text form carries no fractional part. Parquet and Excel keep the real
|
|
88
|
+
Datetime values (millisecond resolution)."""
|
|
89
|
+
datetime_cols = [
|
|
90
|
+
c for c, dtype in df.schema.items() if isinstance(dtype, pl.Datetime)
|
|
91
|
+
]
|
|
92
|
+
if not datetime_cols:
|
|
93
|
+
return df
|
|
94
|
+
return df.with_columns(
|
|
95
|
+
pl.col(c).dt.truncate("1s").dt.to_string("%Y-%m-%dT%H:%M:%S")
|
|
96
|
+
for c in datetime_cols
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
82
100
|
def write_table(
|
|
83
101
|
df: pl.DataFrame,
|
|
84
102
|
path: Path,
|
|
@@ -86,11 +104,11 @@ def write_table(
|
|
|
86
104
|
sheet_name: str = "data",
|
|
87
105
|
) -> None:
|
|
88
106
|
if fmt == "csv":
|
|
89
|
-
df.write_csv(path)
|
|
107
|
+
_second_resolution(df).write_csv(path)
|
|
90
108
|
elif fmt == "tsv":
|
|
91
|
-
df.write_csv(path, separator="\t")
|
|
109
|
+
_second_resolution(df).write_csv(path, separator="\t")
|
|
92
110
|
elif fmt == "jsonl":
|
|
93
|
-
df.write_ndjson(path)
|
|
111
|
+
_second_resolution(df).write_ndjson(path)
|
|
94
112
|
elif fmt == "parquet":
|
|
95
113
|
df.write_parquet(path)
|
|
96
114
|
elif fmt == "xlsx":
|
|
@@ -104,21 +122,21 @@ def write_table(
|
|
|
104
122
|
SCRIPT_SUFFIX = ".ntd"
|
|
105
123
|
|
|
106
124
|
|
|
125
|
+
def step_line(key: str, params: dict) -> str:
|
|
126
|
+
"""One operation description: `<key>(<param>=<json value>, ...)`."""
|
|
127
|
+
return f"{key}({', '.join(f'{p}={json.dumps(v)}' for p, v in params.items())})"
|
|
128
|
+
|
|
129
|
+
|
|
107
130
|
def script_text(steps: list[tuple[str, dict]], header_fields: dict[str, str]) -> str:
|
|
108
131
|
"""Human-readable ASCII text, one operation description per line.
|
|
109
132
|
|
|
110
|
-
Every operation is `<key>(<param>=<json value>, ...)`
|
|
111
|
-
readable while remaining mechanically parseable for a
|
|
112
|
-
"apply a script" capability. `header_fields` become leading `#`
|
|
133
|
+
Every operation is `step_line()`'s `<key>(<param>=<json value>, ...)`
|
|
134
|
+
so the lines stay readable while remaining mechanically parseable for a
|
|
135
|
+
future "apply a script" capability. `header_fields` become leading `#`
|
|
113
136
|
comment lines (source file, save time, ...)"""
|
|
114
137
|
lines = ["# normalize-tabular-data script"]
|
|
115
138
|
lines += [f"# {text}" for text in header_fields.values() if text]
|
|
116
|
-
lines += [
|
|
117
|
-
f"{key}("
|
|
118
|
-
+ ", ".join(f"{param}={json.dumps(value)}" for param, value in params.items())
|
|
119
|
-
+ ")"
|
|
120
|
-
for key, params in steps
|
|
121
|
-
]
|
|
139
|
+
lines += [step_line(key, params) for key, params in steps]
|
|
122
140
|
return "\n".join(lines) + "\n"
|
|
123
141
|
|
|
124
142
|
|
{normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/ops.py
RENAMED
|
@@ -48,8 +48,20 @@ def _apply_date_normalize(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
|
48
48
|
series = df.get_column(col)
|
|
49
49
|
if series.dtype != pl.String:
|
|
50
50
|
series = series.cast(pl.String)
|
|
51
|
-
# Datetime
|
|
52
|
-
return df.with_columns(
|
|
51
|
+
# canonical UTC Datetime at millisecond resolution; unparseable -> null
|
|
52
|
+
return df.with_columns(
|
|
53
|
+
date_parser.parse_series(series).dt.cast_time_unit("ms").alias(col)
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _apply_date_only(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
58
|
+
col: str = p["column"]
|
|
59
|
+
series = df.get_column(col)
|
|
60
|
+
if series.dtype != pl.String:
|
|
61
|
+
series = series.cast(pl.String)
|
|
62
|
+
# date-only ISO-8601: any input format condensed to the UTC calendar
|
|
63
|
+
# date; unparseable values become null
|
|
64
|
+
return df.with_columns(date_parser.parse_series(series).dt.date().alias(col))
|
|
53
65
|
|
|
54
66
|
|
|
55
67
|
def _apply_trim_collapse(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
@@ -117,8 +129,8 @@ def _apply_split_column(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
|
117
129
|
OPS: tuple[Operation, ...] = (
|
|
118
130
|
Operation(
|
|
119
131
|
key="date_normalize",
|
|
120
|
-
title="Normalize
|
|
121
|
-
hotkey="
|
|
132
|
+
title="Normalize date(t)imes",
|
|
133
|
+
hotkey="t",
|
|
122
134
|
params=(
|
|
123
135
|
ParamSpec(
|
|
124
136
|
"column",
|
|
@@ -129,6 +141,20 @@ OPS: tuple[Operation, ...] = (
|
|
|
129
141
|
),
|
|
130
142
|
apply=_apply_date_normalize,
|
|
131
143
|
),
|
|
144
|
+
Operation(
|
|
145
|
+
key="date_only",
|
|
146
|
+
title="Normalize (d)ates",
|
|
147
|
+
hotkey="d",
|
|
148
|
+
params=(
|
|
149
|
+
ParamSpec(
|
|
150
|
+
"column",
|
|
151
|
+
"column",
|
|
152
|
+
"Date column",
|
|
153
|
+
help="Any input format; condenses to the UTC calendar date",
|
|
154
|
+
),
|
|
155
|
+
),
|
|
156
|
+
apply=_apply_date_only,
|
|
157
|
+
),
|
|
132
158
|
Operation(
|
|
133
159
|
key="trim_collapse",
|
|
134
160
|
title="Trim whitespace",
|
|
@@ -143,22 +169,6 @@ OPS: tuple[Operation, ...] = (
|
|
|
143
169
|
),
|
|
144
170
|
apply=_apply_trim_collapse,
|
|
145
171
|
),
|
|
146
|
-
Operation(
|
|
147
|
-
key="dedup_rows",
|
|
148
|
-
title="Deduplicate rows",
|
|
149
|
-
hotkey="p",
|
|
150
|
-
params=(
|
|
151
|
-
ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
|
|
152
|
-
ParamSpec(
|
|
153
|
-
"keep",
|
|
154
|
-
"choice",
|
|
155
|
-
"Keep",
|
|
156
|
-
default="first",
|
|
157
|
-
choices=("first", "last"),
|
|
158
|
-
),
|
|
159
|
-
),
|
|
160
|
-
apply=_apply_dedup_rows,
|
|
161
|
-
),
|
|
162
172
|
Operation(
|
|
163
173
|
key="combine_columns",
|
|
164
174
|
title="Combine columns",
|
|
@@ -203,6 +213,22 @@ OPS: tuple[Operation, ...] = (
|
|
|
203
213
|
),
|
|
204
214
|
apply=_apply_drop_columns,
|
|
205
215
|
),
|
|
216
|
+
Operation(
|
|
217
|
+
key="dedup_rows",
|
|
218
|
+
title="Deduplicate rows",
|
|
219
|
+
hotkey="p",
|
|
220
|
+
params=(
|
|
221
|
+
ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
|
|
222
|
+
ParamSpec(
|
|
223
|
+
"keep",
|
|
224
|
+
"choice",
|
|
225
|
+
"Keep",
|
|
226
|
+
default="first",
|
|
227
|
+
choices=("first", "last"),
|
|
228
|
+
),
|
|
229
|
+
),
|
|
230
|
+
apply=_apply_dedup_rows,
|
|
231
|
+
),
|
|
206
232
|
)
|
|
207
233
|
|
|
208
234
|
OP_REGISTRY: dict[str, Operation] = {op.key: op for op in OPS}
|
|
@@ -224,6 +250,47 @@ RENAME_OP = Operation(
|
|
|
224
250
|
PLAY_REGISTRY: dict[str, Operation] = {**OP_REGISTRY, RENAME_OP.key: RENAME_OP}
|
|
225
251
|
|
|
226
252
|
|
|
253
|
+
def apply_script_steps(
|
|
254
|
+
steps: list[tuple[str, dict]],
|
|
255
|
+
pipeline: Pipeline,
|
|
256
|
+
op_log: list[tuple[str, Any]],
|
|
257
|
+
on_failure: Callable[[str], None],
|
|
258
|
+
) -> bool:
|
|
259
|
+
"""Apply a played script's steps to `pipeline`, all-or-nothing.
|
|
260
|
+
|
|
261
|
+
Each step is resolved against PLAY_REGISTRY, applied, and logged to
|
|
262
|
+
`op_log` (the same keys the interactive player records). On the first
|
|
263
|
+
step that is impossible for the currently loaded file the work done so
|
|
264
|
+
far is rolled back — `pipeline.applied` and `op_log` truncated to where
|
|
265
|
+
the script started and the redo stack cleared — and the failure reason
|
|
266
|
+
is passed to `on_failure`. Returns False when a step failed, in which
|
|
267
|
+
case `pipeline` is left exactly as it was before the call."""
|
|
268
|
+
started_applied = len(pipeline.applied)
|
|
269
|
+
started_log = len(op_log)
|
|
270
|
+
|
|
271
|
+
def give_up(reason: str) -> None:
|
|
272
|
+
del pipeline.applied[started_applied:]
|
|
273
|
+
pipeline.redo_stack.clear()
|
|
274
|
+
del op_log[started_log:]
|
|
275
|
+
on_failure(reason)
|
|
276
|
+
|
|
277
|
+
for index, (key, params) in enumerate(steps, 1):
|
|
278
|
+
op = PLAY_REGISTRY.get(key)
|
|
279
|
+
if op is None:
|
|
280
|
+
give_up(f"Step {index}: {key!r} is not a known operation")
|
|
281
|
+
return False
|
|
282
|
+
try:
|
|
283
|
+
pipeline.apply(op, params)
|
|
284
|
+
# the pipeline refolds lazily; fold now so an op-specific error
|
|
285
|
+
# (e.g. unparseable dates) rolls back this step too
|
|
286
|
+
pipeline.current()
|
|
287
|
+
except Exception as exc:
|
|
288
|
+
give_up(f"Step {index} of {len(steps)} ({op.title}) failed: {exc}")
|
|
289
|
+
return False
|
|
290
|
+
op_log.append((key, dict(params)))
|
|
291
|
+
return True
|
|
292
|
+
|
|
293
|
+
|
|
227
294
|
# --- pipeline ---------------------------------------------------------------
|
|
228
295
|
|
|
229
296
|
|
{normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/screens.py
RENAMED
|
@@ -363,7 +363,11 @@ class OpParamsModal(ModalDialog):
|
|
|
363
363
|
|
|
364
364
|
def compose_body(self) -> ComposeResult:
|
|
365
365
|
self.widgets: dict[str, object] = {}
|
|
366
|
-
candidates =
|
|
366
|
+
candidates = (
|
|
367
|
+
self._date_candidates()
|
|
368
|
+
if self.op.key in ("date_normalize", "date_only")
|
|
369
|
+
else []
|
|
370
|
+
)
|
|
367
371
|
for spec in self.op.params:
|
|
368
372
|
if spec.kind == "column_multi":
|
|
369
373
|
sel = SelectionList(
|
|
@@ -531,8 +535,8 @@ class SaveModal(ModalDialog):
|
|
|
531
535
|
class ScriptNameModal(ModalDialog):
|
|
532
536
|
"""Name the '.ntd' script file when saving a sequence of operations.
|
|
533
537
|
|
|
534
|
-
The
|
|
535
|
-
|
|
538
|
+
The extension is always .ntd: whatever the user types, the saved name
|
|
539
|
+
ends in the fixed script suffix."""
|
|
536
540
|
|
|
537
541
|
dialog_title = "Save script"
|
|
538
542
|
|
|
@@ -547,17 +551,14 @@ class ScriptNameModal(ModalDialog):
|
|
|
547
551
|
id="script_input",
|
|
548
552
|
)
|
|
549
553
|
yield self.path_input
|
|
550
|
-
yield Static(
|
|
551
|
-
f"Default extension: {SCRIPT_SUFFIX} — type another to override",
|
|
552
|
-
classes="help",
|
|
553
|
-
)
|
|
554
|
+
yield Static(f"Scripts always end in {SCRIPT_SUFFIX}", classes="help")
|
|
554
555
|
|
|
555
556
|
def action_ok(self) -> None:
|
|
556
557
|
raw = self.path_input.value.strip()
|
|
557
558
|
if not raw:
|
|
558
559
|
self.app.notify("Type a script file name or path", severity="error")
|
|
559
560
|
return
|
|
560
|
-
target = Path(raw).expanduser()
|
|
561
|
+
target = Path(raw).expanduser().with_suffix(SCRIPT_SUFFIX)
|
|
561
562
|
if target.exists():
|
|
562
563
|
self.app.notify("Script file exists: pick another name", severity="warning")
|
|
563
564
|
return
|
{normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/widgets.py
RENAMED
|
@@ -92,13 +92,13 @@ class ColumnSidebar(Widget):
|
|
|
92
92
|
|
|
93
93
|
|
|
94
94
|
def _dtype_mark(dtype: str) -> str:
|
|
95
|
-
"""Compact 3-wide dtype mark for the sidebar: dt (datetime
|
|
96
|
-
str (string), int (integer), dec (float/decimal),
|
|
97
|
-
everything else oth. Each mark is padded to 3 columns
|
|
98
|
-
names left-align."""
|
|
95
|
+
"""Compact 3-wide dtype mark for the sidebar: dt (datetime),
|
|
96
|
+
day (date-only), str (string), int (integer), dec (float/decimal),
|
|
97
|
+
t/f (boolean); everything else oth. Each mark is padded to 3 columns
|
|
98
|
+
so column names left-align."""
|
|
99
99
|
for prefix, mark in (
|
|
100
100
|
("Datetime", "dt "),
|
|
101
|
-
("Date", "
|
|
101
|
+
("Date", "day"),
|
|
102
102
|
("Time", "dt "),
|
|
103
103
|
("String", "str"),
|
|
104
104
|
("Categorical", "str"),
|
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
"""normalize-tabular-data: TUI for normalizing tabular data with polars."""
|
|
2
|
-
|
|
3
|
-
__version__ = "0.1.0"
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
def main() -> None:
|
|
7
|
-
try:
|
|
8
|
-
from normalize_tabular_data.app import NormalizeApp
|
|
9
|
-
except ImportError as exc: # helpful message if an optional engine is missing
|
|
10
|
-
import sys
|
|
11
|
-
|
|
12
|
-
print(
|
|
13
|
-
f"normalize-tabular-data is missing a dependency ({exc}).\n"
|
|
14
|
-
"Reinstall with: uv tool install --force --reinstall normalize-tabular-data",
|
|
15
|
-
file=sys.stderr,
|
|
16
|
-
)
|
|
17
|
-
raise SystemExit(1)
|
|
18
|
-
NormalizeApp().run()
|
{normalize_tabular_data-0.1.1 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/__main__.py
RENAMED
|
File without changes
|