normalize-tabular-data 0.1.2__tar.gz → 0.1.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/LICENSE +1 -1
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/PKG-INFO +53 -64
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/README.md +50 -62
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/pyproject.toml +4 -2
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/pyproject.toml.orig +4 -2
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/__init__.py +1 -1
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/app.py +64 -3
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/io.py +21 -3
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/ops.py +44 -27
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/screens.py +9 -8
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/widgets.py +5 -5
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/__main__.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: normalize-tabular-data
|
|
3
|
-
Version: 0.1.
|
|
4
|
-
Summary: TUI for normalizing tabular data with
|
|
3
|
+
Version: 0.1.4
|
|
4
|
+
Summary: TUI for normalizing tabular data with Polars
|
|
5
5
|
Author: David Mertz, Ph.D.
|
|
6
6
|
Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
|
|
7
7
|
License-Expression: BSD-2-Clause
|
|
@@ -11,58 +11,81 @@ Requires-Dist: textual>=8.2.8
|
|
|
11
11
|
Requires-Dist: polars>=1.44,<2
|
|
12
12
|
Requires-Dist: fastexcel>=0.10
|
|
13
13
|
Requires-Dist: xlsxwriter>=0.9
|
|
14
|
+
Requires-Dist: platformdirs>=3.0
|
|
14
15
|
Requires-Python: >=3.14
|
|
15
16
|
Description-Content-Type: text/markdown
|
|
16
17
|
|
|
17
|
-
# normalize-tabular-data
|
|
18
|
+
# normalize-tabular-data (NTD)
|
|
18
19
|
|
|
19
20
|
A terminal UI (Textual) for interactively normalizing tabular data, backed by
|
|
20
21
|
[polars](https://pola.rs) for speed and
|
|
21
22
|
[gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
|
|
22
23
|
date parsing at Rust speed.
|
|
23
24
|
|
|
25
|
+
NTD also supports purely command-line operations for use in scripting, launched
|
|
26
|
+
with `uvx` with no other requirements that are not dynamically downloaded.
|
|
27
|
+
|
|
24
28
|
Load a file, see a preview, build up a pipeline of normalization operations
|
|
25
29
|
(normalize messy dates, trim whitespace, rename a column, deduplicate,
|
|
26
|
-
combine/split columns, drop columns), then save the
|
|
27
|
-
cleaned result.
|
|
30
|
+
combine/split columns, drop columns), then save the cleaned result.
|
|
28
31
|
|
|
29
|
-
|
|
32
|
+
## Formats
|
|
30
33
|
|
|
31
|
-
|
|
34
|
+
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
35
|
+
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
32
36
|
|
|
33
|
-
|
|
37
|
+
Large files are previewed with a random sample of 250 rows; every operation
|
|
38
|
+
still runs on the full data.
|
|
34
39
|
|
|
35
|
-
|
|
36
|
-
uv tool install normalize-tabular-data
|
|
37
|
-
# upgrade later with:
|
|
38
|
-
uv tool upgrade normalize-tabular-data
|
|
39
|
-
```
|
|
40
|
+
## Running the Text User Interface
|
|
40
41
|
|
|
41
|
-
|
|
42
|
+

|
|
42
43
|
|
|
43
|
-
|
|
44
|
+
Installation and launch options — persistent install, ephemeral `uvx`
|
|
45
|
+
runs, running from a git checkout, the single-file launchers for
|
|
46
|
+
Linux/macOS/Windows, and installing them as desktop icons — are
|
|
47
|
+
described in [docs/running-the-tool.md](docs/running-the-tool.md).
|
|
44
48
|
|
|
45
|
-
|
|
46
|
-
uvx normalize-tabular-data
|
|
47
|
-
# pin a specific version:
|
|
48
|
-
uvx normalize-tabular-data==0.1.1
|
|
49
|
-
```
|
|
49
|
+
## Capabilities in the TUI
|
|
50
50
|
|
|
51
|
-
|
|
51
|
+
Launch with `uvx normalize-tabular-data` (or a launcher). Keys:
|
|
52
52
|
|
|
53
|
-
Within the directory of the cloned repository:
|
|
54
53
|
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
54
|
+
| Key | Action |
|
|
55
|
+
|-----|------------------------------|
|
|
56
|
+
| `f` | (F)ile — open a file |
|
|
57
|
+
| `o` | (O)peration — choose an op |
|
|
58
|
+
| `p` | (P)lay script — apply `.ntd` |
|
|
59
|
+
| `u` | (U)ndo last applied step |
|
|
60
|
+
| `r` | (R)edo |
|
|
61
|
+
| `s` | (S)ave the data in a format |
|
|
62
|
+
| `q` | (Q)uit |
|
|
63
|
+
|
|
64
|
+
## Operations
|
|
58
65
|
|
|
59
|
-
|
|
66
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
67
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
68
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
69
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
70
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
71
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
72
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
73
|
+
(`year-month-day`); unparseable values become null.
|
|
74
|
+
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
75
|
+
per selected columns.
|
|
76
|
+
- Rename a column — click its header in the preview and type the new name.
|
|
77
|
+
- **Combine columns** — concatenate two or more columns with a separator.
|
|
78
|
+
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
79
|
+
delimiter splits on runs of whitespace.
|
|
80
|
+
- **Remove columns** — drop selected columns entirely.
|
|
81
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
82
|
+
|
|
83
|
+
## Command line usage
|
|
60
84
|
|
|
61
85
|
An optional `FILE` argument names a table to open at startup, as if you
|
|
62
86
|
had picked it from the in-app dialog.
|
|
63
87
|
|
|
64
88
|
```bash
|
|
65
|
-
normalize-tabular-data data/members.csv # or with uvx/uv run
|
|
66
89
|
uvx normalize-tabular-data data/members.tsv
|
|
67
90
|
```
|
|
68
91
|
|
|
@@ -103,48 +126,14 @@ to stdout itself and the step lines are suppressed, so stdout carries
|
|
|
103
126
|
only the data. After a successful run the confirmation line goes to
|
|
104
127
|
stderr.
|
|
105
128
|
|
|
106
|
-
## Usage
|
|
107
|
-
|
|
108
|
-
Launch with `normalize-tabular-data`. Keys:
|
|
109
|
-
|
|
110
|
-
| Key | Action |
|
|
111
|
-
|-----|------------------------------|
|
|
112
|
-
| `f` | (F)ile — open a file |
|
|
113
|
-
| `o` | (O)peration — choose an op |
|
|
114
|
-
| `p` | (P)lay script — apply `.ntd` |
|
|
115
|
-
| `u` | (U)ndo last applied step |
|
|
116
|
-
| `r` | (R)edo |
|
|
117
|
-
| `s` | (S)ave the data in a format |
|
|
118
|
-
| `q` | (Q)uit |
|
|
119
|
-
|
|
120
|
-
## Operations
|
|
121
|
-
|
|
122
|
-
- **Normalize dates** — parse a messy date column of *any* input format into
|
|
123
|
-
canonical UTC `year-month-dayThh:mm:ss` values (second resolution;
|
|
124
|
-
fractional parts are truncated); unparseable values become null.
|
|
125
|
-
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
126
|
-
per selected columns.
|
|
127
|
-
- Rename a column — click its header in the preview and type the new name.
|
|
128
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
129
|
-
- **Combine columns** — concatenate two or more columns with a separator.
|
|
130
|
-
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
131
|
-
delimiter splits on runs of whitespace.
|
|
132
|
-
- **Remove columns** — drop selected columns entirely.
|
|
133
|
-
|
|
134
|
-
## Formats
|
|
135
|
-
|
|
136
|
-
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
137
|
-
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
138
|
-
|
|
139
|
-
Large files are previewed with a random sample of 250 rows; every operation
|
|
140
|
-
still runs on the full data.
|
|
141
|
-
|
|
142
129
|
## Operation scripts (`.ntd`)
|
|
143
130
|
|
|
144
131
|
Every open file keeps an internal log of the operations performed on it. When
|
|
145
132
|
saving, the dialog offers a **Save sequence of operations?** checkbox
|
|
146
133
|
(unticked by default); with it ticked, a second dialog asks where to write
|
|
147
|
-
the script — suggested `<table>.ntd
|
|
134
|
+
the script — suggested `<table>.ntd`; scripts always use the `.ntd`
|
|
135
|
+
extension, so the name you type is saved with that suffix whatever it
|
|
136
|
+
ends in.
|
|
148
137
|
|
|
149
138
|
The script is plain ASCII text, one operation per line, e.g.:
|
|
150
139
|
|
|
@@ -1,52 +1,74 @@
|
|
|
1
|
-
# normalize-tabular-data
|
|
1
|
+
# normalize-tabular-data (NTD)
|
|
2
2
|
|
|
3
3
|
A terminal UI (Textual) for interactively normalizing tabular data, backed by
|
|
4
4
|
[polars](https://pola.rs) for speed and
|
|
5
5
|
[gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
|
|
6
6
|
date parsing at Rust speed.
|
|
7
7
|
|
|
8
|
+
NTD also supports purely command-line operations for use in scripting, launched
|
|
9
|
+
with `uvx` with no other requirements that are not dynamically downloaded.
|
|
10
|
+
|
|
8
11
|
Load a file, see a preview, build up a pipeline of normalization operations
|
|
9
12
|
(normalize messy dates, trim whitespace, rename a column, deduplicate,
|
|
10
|
-
combine/split columns, drop columns), then save the
|
|
11
|
-
cleaned result.
|
|
13
|
+
combine/split columns, drop columns), then save the cleaned result.
|
|
12
14
|
|
|
13
|
-
|
|
15
|
+
## Formats
|
|
14
16
|
|
|
15
|
-
|
|
17
|
+
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
18
|
+
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
16
19
|
|
|
17
|
-
|
|
20
|
+
Large files are previewed with a random sample of 250 rows; every operation
|
|
21
|
+
still runs on the full data.
|
|
18
22
|
|
|
19
|
-
|
|
20
|
-
uv tool install normalize-tabular-data
|
|
21
|
-
# upgrade later with:
|
|
22
|
-
uv tool upgrade normalize-tabular-data
|
|
23
|
-
```
|
|
23
|
+
## Running the Text User Interface
|
|
24
24
|
|
|
25
|
-
|
|
25
|
+

|
|
26
26
|
|
|
27
|
-
|
|
27
|
+
Installation and launch options — persistent install, ephemeral `uvx`
|
|
28
|
+
runs, running from a git checkout, the single-file launchers for
|
|
29
|
+
Linux/macOS/Windows, and installing them as desktop icons — are
|
|
30
|
+
described in [docs/running-the-tool.md](docs/running-the-tool.md).
|
|
28
31
|
|
|
29
|
-
|
|
30
|
-
uvx normalize-tabular-data
|
|
31
|
-
# pin a specific version:
|
|
32
|
-
uvx normalize-tabular-data==0.1.1
|
|
33
|
-
```
|
|
32
|
+
## Capabilities in the TUI
|
|
34
33
|
|
|
35
|
-
|
|
34
|
+
Launch with `uvx normalize-tabular-data` (or a launcher). Keys:
|
|
36
35
|
|
|
37
|
-
Within the directory of the cloned repository:
|
|
38
36
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
37
|
+
| Key | Action |
|
|
38
|
+
|-----|------------------------------|
|
|
39
|
+
| `f` | (F)ile — open a file |
|
|
40
|
+
| `o` | (O)peration — choose an op |
|
|
41
|
+
| `p` | (P)lay script — apply `.ntd` |
|
|
42
|
+
| `u` | (U)ndo last applied step |
|
|
43
|
+
| `r` | (R)edo |
|
|
44
|
+
| `s` | (S)ave the data in a format |
|
|
45
|
+
| `q` | (Q)uit |
|
|
46
|
+
|
|
47
|
+
## Operations
|
|
42
48
|
|
|
43
|
-
|
|
49
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
50
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
51
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
52
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
53
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
54
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
55
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
56
|
+
(`year-month-day`); unparseable values become null.
|
|
57
|
+
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
58
|
+
per selected columns.
|
|
59
|
+
- Rename a column — click its header in the preview and type the new name.
|
|
60
|
+
- **Combine columns** — concatenate two or more columns with a separator.
|
|
61
|
+
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
62
|
+
delimiter splits on runs of whitespace.
|
|
63
|
+
- **Remove columns** — drop selected columns entirely.
|
|
64
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
65
|
+
|
|
66
|
+
## Command line usage
|
|
44
67
|
|
|
45
68
|
An optional `FILE` argument names a table to open at startup, as if you
|
|
46
69
|
had picked it from the in-app dialog.
|
|
47
70
|
|
|
48
71
|
```bash
|
|
49
|
-
normalize-tabular-data data/members.csv # or with uvx/uv run
|
|
50
72
|
uvx normalize-tabular-data data/members.tsv
|
|
51
73
|
```
|
|
52
74
|
|
|
@@ -87,48 +109,14 @@ to stdout itself and the step lines are suppressed, so stdout carries
|
|
|
87
109
|
only the data. After a successful run the confirmation line goes to
|
|
88
110
|
stderr.
|
|
89
111
|
|
|
90
|
-
## Usage
|
|
91
|
-
|
|
92
|
-
Launch with `normalize-tabular-data`. Keys:
|
|
93
|
-
|
|
94
|
-
| Key | Action |
|
|
95
|
-
|-----|------------------------------|
|
|
96
|
-
| `f` | (F)ile — open a file |
|
|
97
|
-
| `o` | (O)peration — choose an op |
|
|
98
|
-
| `p` | (P)lay script — apply `.ntd` |
|
|
99
|
-
| `u` | (U)ndo last applied step |
|
|
100
|
-
| `r` | (R)edo |
|
|
101
|
-
| `s` | (S)ave the data in a format |
|
|
102
|
-
| `q` | (Q)uit |
|
|
103
|
-
|
|
104
|
-
## Operations
|
|
105
|
-
|
|
106
|
-
- **Normalize dates** — parse a messy date column of *any* input format into
|
|
107
|
-
canonical UTC `year-month-dayThh:mm:ss` values (second resolution;
|
|
108
|
-
fractional parts are truncated); unparseable values become null.
|
|
109
|
-
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
110
|
-
per selected columns.
|
|
111
|
-
- Rename a column — click its header in the preview and type the new name.
|
|
112
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
113
|
-
- **Combine columns** — concatenate two or more columns with a separator.
|
|
114
|
-
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
115
|
-
delimiter splits on runs of whitespace.
|
|
116
|
-
- **Remove columns** — drop selected columns entirely.
|
|
117
|
-
|
|
118
|
-
## Formats
|
|
119
|
-
|
|
120
|
-
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
121
|
-
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
122
|
-
|
|
123
|
-
Large files are previewed with a random sample of 250 rows; every operation
|
|
124
|
-
still runs on the full data.
|
|
125
|
-
|
|
126
112
|
## Operation scripts (`.ntd`)
|
|
127
113
|
|
|
128
114
|
Every open file keeps an internal log of the operations performed on it. When
|
|
129
115
|
saving, the dialog offers a **Save sequence of operations?** checkbox
|
|
130
116
|
(unticked by default); with it ticked, a second dialog asks where to write
|
|
131
|
-
the script — suggested `<table>.ntd
|
|
117
|
+
the script — suggested `<table>.ntd`; scripts always use the `.ntd`
|
|
118
|
+
extension, so the name you type is saved with that suffix whatever it
|
|
119
|
+
ends in.
|
|
132
120
|
|
|
133
121
|
The script is plain ASCII text, one operation per line, e.g.:
|
|
134
122
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.1.
|
|
4
|
-
description = "TUI for normalizing tabular data with
|
|
3
|
+
version = "0.1.4"
|
|
4
|
+
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
@@ -12,6 +12,7 @@ dependencies = [
|
|
|
12
12
|
"polars>=1.44,<2",
|
|
13
13
|
"fastexcel>=0.10",
|
|
14
14
|
"xlsxwriter>=0.9",
|
|
15
|
+
"platformdirs>=3.0",
|
|
15
16
|
]
|
|
16
17
|
|
|
17
18
|
[[project.authors]]
|
|
@@ -27,6 +28,7 @@ dev = [
|
|
|
27
28
|
"pytest-asyncio>=0.25",
|
|
28
29
|
"ruff>=0.16.10",
|
|
29
30
|
"twine>=7.0.0",
|
|
31
|
+
"ty>=0.0.84",
|
|
30
32
|
]
|
|
31
33
|
|
|
32
34
|
[build-system]
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.1.
|
|
4
|
-
description = "TUI for normalizing tabular data with
|
|
3
|
+
version = "0.1.4"
|
|
4
|
+
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
@@ -15,6 +15,7 @@ dependencies = [
|
|
|
15
15
|
"polars>=1.44,<2",
|
|
16
16
|
"fastexcel>=0.10",
|
|
17
17
|
"xlsxwriter>=0.9",
|
|
18
|
+
"platformdirs>=3.0",
|
|
18
19
|
]
|
|
19
20
|
|
|
20
21
|
[dependency-groups]
|
|
@@ -23,6 +24,7 @@ dev = [
|
|
|
23
24
|
"pytest-asyncio>=0.25",
|
|
24
25
|
"ruff>=0.16.10",
|
|
25
26
|
"twine>=7.0.0",
|
|
27
|
+
"ty>=0.0.84",
|
|
26
28
|
]
|
|
27
29
|
|
|
28
30
|
[project.scripts]
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/app.py
RENAMED
|
@@ -6,6 +6,7 @@ import bisect
|
|
|
6
6
|
import datetime as _dt
|
|
7
7
|
import random
|
|
8
8
|
from pathlib import Path
|
|
9
|
+
from platformdirs import user_config_dir
|
|
9
10
|
from typing import Any, Iterable
|
|
10
11
|
|
|
11
12
|
from textual import on
|
|
@@ -43,6 +44,7 @@ from normalize_tabular_data.screens import (
|
|
|
43
44
|
from normalize_tabular_data.widgets import ColumnSidebar, MenuFooter, StepsBar
|
|
44
45
|
|
|
45
46
|
PREVIEW_ROWS = 250
|
|
47
|
+
APP_CONFIG_NAME = "normalize-tabular-data"
|
|
46
48
|
|
|
47
49
|
|
|
48
50
|
class CommandMenu(CommandPalette):
|
|
@@ -135,10 +137,19 @@ class OpChooserModal(ModalDialog):
|
|
|
135
137
|
|
|
136
138
|
def _marked_title(self, op_key: str, title: str, hotkey: str = "") -> str:
|
|
137
139
|
"""Return "Normalize (d)ates"-style markup: the op's hotkey letter
|
|
138
|
-
highlighted where it sits in the title (keeping its case).
|
|
139
|
-
|
|
140
|
-
|
|
140
|
+
highlighted where it sits in the title (keeping its case). A title
|
|
141
|
+
that already parenthesizes its letter ("Normalize date(t)imes")
|
|
142
|
+
highlights the letter between the parens; a plain title matches
|
|
143
|
+
its first occurrence and adds the parens. Ops without a designated
|
|
144
|
+
hotkey claim the first unused letter in the title instead; plain
|
|
145
|
+
title if no letter can be claimed."""
|
|
141
146
|
if hotkey:
|
|
147
|
+
i = title.lower().find(f"({hotkey.lower()})")
|
|
148
|
+
if i >= 0:
|
|
149
|
+
# the letter to claim sits inside existing parentheses:
|
|
150
|
+
# highlight it without adding a second pair
|
|
151
|
+
self._hotkeys[title[i + 1].lower()] = op_key
|
|
152
|
+
return f"{title[: i + 1]}[cyan]{title[i + 1]}[/cyan]{title[i + 2:]}"
|
|
142
153
|
i = title.lower().find(hotkey.lower())
|
|
143
154
|
else:
|
|
144
155
|
i = next(
|
|
@@ -275,8 +286,15 @@ class NormalizeApp(App[None]):
|
|
|
275
286
|
self,
|
|
276
287
|
initial_file: str | Path | None = None,
|
|
277
288
|
initial_script: str | Path | None = None,
|
|
289
|
+
config_dir: Path | None = None,
|
|
278
290
|
) -> None:
|
|
279
291
|
super().__init__()
|
|
292
|
+
# where the chosen theme (and any future settings) are stored;
|
|
293
|
+
# None means the per-user platform directory
|
|
294
|
+
self._config_dir = config_dir
|
|
295
|
+
# flips to True at on_mount: only theme changes made after are the
|
|
296
|
+
# user's and only those are persisted
|
|
297
|
+
self._app_ready = False
|
|
280
298
|
# filesystem path to open when the app starts (the optional file
|
|
281
299
|
# named on the command line); loaded like a file chosen in-app
|
|
282
300
|
self.initial_file: Path | None = (
|
|
@@ -354,6 +372,9 @@ class NormalizeApp(App[None]):
|
|
|
354
372
|
return list(self.pipeline.current().columns)
|
|
355
373
|
|
|
356
374
|
def on_mount(self) -> None:
|
|
375
|
+
# remember this point so theme changes from here on are the
|
|
376
|
+
# user's own choices and get persisted
|
|
377
|
+
self._apply_saved_theme()
|
|
357
378
|
self.push_screen(MainScreen())
|
|
358
379
|
self._refresh_steps()
|
|
359
380
|
# the command line's file, if any, is opened once the main screen
|
|
@@ -361,6 +382,46 @@ class NormalizeApp(App[None]):
|
|
|
361
382
|
# crashing if it is unreadable or has an unsupported format
|
|
362
383
|
if self.initial_file is not None:
|
|
363
384
|
self.call_after_refresh(self.load_path, self.initial_file)
|
|
385
|
+
self._app_ready = True
|
|
386
|
+
|
|
387
|
+
# --- theme persistence ---------------------------------------------
|
|
388
|
+
|
|
389
|
+
def watch_theme(self, theme_name: str) -> None:
|
|
390
|
+
"""Persist every theme the user chooses (menu or command), so the
|
|
391
|
+
next launch opens with it again. Skipped while headless, so the
|
|
392
|
+
pilot tests cannot touch the real configuration; skipped before
|
|
393
|
+
mount, where only the reactive's default applies."""
|
|
394
|
+
if self._app_ready and not self.is_headless:
|
|
395
|
+
try:
|
|
396
|
+
self._theme_config_path().parent.mkdir(parents=True, exist_ok=True)
|
|
397
|
+
self._theme_config_path().write_text(
|
|
398
|
+
theme_name + "\n", encoding="utf-8"
|
|
399
|
+
)
|
|
400
|
+
except OSError:
|
|
401
|
+
pass # unwritable config location: run without remembering
|
|
402
|
+
|
|
403
|
+
def _apply_saved_theme(self) -> None:
|
|
404
|
+
"""Restore the theme chosen in a previous session, if it still exists.
|
|
405
|
+
|
|
406
|
+
Skipped in headless runs that use the real configuration location
|
|
407
|
+
(pilot tests): an eyeballing developer's own saved theme should
|
|
408
|
+
not leak into supposedly default-themed tests. A headless run
|
|
409
|
+
given an explicit config_dir is a deliberate fixture and loads it."""
|
|
410
|
+
if self.is_headless and self._config_dir is None:
|
|
411
|
+
return
|
|
412
|
+
try:
|
|
413
|
+
saved = self._theme_config_path().read_text(encoding="utf-8").strip()
|
|
414
|
+
except OSError:
|
|
415
|
+
return
|
|
416
|
+
if not saved:
|
|
417
|
+
return
|
|
418
|
+
try:
|
|
419
|
+
self.theme = saved
|
|
420
|
+
except Exception:
|
|
421
|
+
pass # a theme name that no longer exists: keep the default
|
|
422
|
+
|
|
423
|
+
def _theme_config_path(self) -> Path:
|
|
424
|
+
return (self._config_dir or Path(user_config_dir(APP_CONFIG_NAME))) / "theme"
|
|
364
425
|
|
|
365
426
|
# --- data plumbing ---------------------------------------------------
|
|
366
427
|
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/io.py
RENAMED
|
@@ -79,6 +79,24 @@ def excel_sheets(path: Path) -> list[str]:
|
|
|
79
79
|
return list(fastexcel.read_excel(path).sheet_names)
|
|
80
80
|
|
|
81
81
|
|
|
82
|
+
def _second_resolution(df: pl.DataFrame) -> pl.DataFrame:
|
|
83
|
+
"""Datetime columns serialized as canonical UTC second-resolution text.
|
|
84
|
+
|
|
85
|
+
Used only for the text formats (CSV/TSV/JSONL): polars pads datetime
|
|
86
|
+
output with sub-second digits no matter the unit, and the canonical
|
|
87
|
+
text form carries no fractional part. Parquet and Excel keep the real
|
|
88
|
+
Datetime values (millisecond resolution)."""
|
|
89
|
+
datetime_cols = [
|
|
90
|
+
c for c, dtype in df.schema.items() if isinstance(dtype, pl.Datetime)
|
|
91
|
+
]
|
|
92
|
+
if not datetime_cols:
|
|
93
|
+
return df
|
|
94
|
+
return df.with_columns(
|
|
95
|
+
pl.col(c).dt.truncate("1s").dt.to_string("%Y-%m-%dT%H:%M:%S")
|
|
96
|
+
for c in datetime_cols
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
82
100
|
def write_table(
|
|
83
101
|
df: pl.DataFrame,
|
|
84
102
|
path: Path,
|
|
@@ -86,11 +104,11 @@ def write_table(
|
|
|
86
104
|
sheet_name: str = "data",
|
|
87
105
|
) -> None:
|
|
88
106
|
if fmt == "csv":
|
|
89
|
-
df.write_csv(path)
|
|
107
|
+
_second_resolution(df).write_csv(path)
|
|
90
108
|
elif fmt == "tsv":
|
|
91
|
-
df.write_csv(path, separator="\t")
|
|
109
|
+
_second_resolution(df).write_csv(path, separator="\t")
|
|
92
110
|
elif fmt == "jsonl":
|
|
93
|
-
df.write_ndjson(path)
|
|
111
|
+
_second_resolution(df).write_ndjson(path)
|
|
94
112
|
elif fmt == "parquet":
|
|
95
113
|
df.write_parquet(path)
|
|
96
114
|
elif fmt == "xlsx":
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/ops.py
RENAMED
|
@@ -48,19 +48,22 @@ def _apply_date_normalize(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
|
48
48
|
series = df.get_column(col)
|
|
49
49
|
if series.dtype != pl.String:
|
|
50
50
|
series = series.cast(pl.String)
|
|
51
|
-
# canonical UTC at
|
|
52
|
-
# Datetime (ns/us/ms only, all of which print sub-second digits in
|
|
53
|
-
# every export), so the normalized column is the truncated
|
|
54
|
-
# `year-month-dayThh:mm:ss` string; fractional parts are dropped and
|
|
55
|
-
# unparseable values become null
|
|
51
|
+
# canonical UTC Datetime at millisecond resolution; unparseable -> null
|
|
56
52
|
return df.with_columns(
|
|
57
|
-
date_parser.parse_series(series)
|
|
58
|
-
.dt.truncate("1s")
|
|
59
|
-
.dt.to_string("%Y-%m-%dT%H:%M:%S")
|
|
60
|
-
.alias(col)
|
|
53
|
+
date_parser.parse_series(series).dt.cast_time_unit("ms").alias(col)
|
|
61
54
|
)
|
|
62
55
|
|
|
63
56
|
|
|
57
|
+
def _apply_date_only(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
58
|
+
col: str = p["column"]
|
|
59
|
+
series = df.get_column(col)
|
|
60
|
+
if series.dtype != pl.String:
|
|
61
|
+
series = series.cast(pl.String)
|
|
62
|
+
# date-only ISO-8601: any input format condensed to the UTC calendar
|
|
63
|
+
# date; unparseable values become null
|
|
64
|
+
return df.with_columns(date_parser.parse_series(series).dt.date().alias(col))
|
|
65
|
+
|
|
66
|
+
|
|
64
67
|
def _apply_trim_collapse(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
65
68
|
cols: list[str] = p["columns"]
|
|
66
69
|
return df.with_columns(
|
|
@@ -126,8 +129,8 @@ def _apply_split_column(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
|
126
129
|
OPS: tuple[Operation, ...] = (
|
|
127
130
|
Operation(
|
|
128
131
|
key="date_normalize",
|
|
129
|
-
title="Normalize
|
|
130
|
-
hotkey="
|
|
132
|
+
title="Normalize date(t)imes",
|
|
133
|
+
hotkey="t",
|
|
131
134
|
params=(
|
|
132
135
|
ParamSpec(
|
|
133
136
|
"column",
|
|
@@ -138,6 +141,20 @@ OPS: tuple[Operation, ...] = (
|
|
|
138
141
|
),
|
|
139
142
|
apply=_apply_date_normalize,
|
|
140
143
|
),
|
|
144
|
+
Operation(
|
|
145
|
+
key="date_only",
|
|
146
|
+
title="Normalize (d)ates",
|
|
147
|
+
hotkey="d",
|
|
148
|
+
params=(
|
|
149
|
+
ParamSpec(
|
|
150
|
+
"column",
|
|
151
|
+
"column",
|
|
152
|
+
"Date column",
|
|
153
|
+
help="Any input format; condenses to the UTC calendar date",
|
|
154
|
+
),
|
|
155
|
+
),
|
|
156
|
+
apply=_apply_date_only,
|
|
157
|
+
),
|
|
141
158
|
Operation(
|
|
142
159
|
key="trim_collapse",
|
|
143
160
|
title="Trim whitespace",
|
|
@@ -152,22 +169,6 @@ OPS: tuple[Operation, ...] = (
|
|
|
152
169
|
),
|
|
153
170
|
apply=_apply_trim_collapse,
|
|
154
171
|
),
|
|
155
|
-
Operation(
|
|
156
|
-
key="dedup_rows",
|
|
157
|
-
title="Deduplicate rows",
|
|
158
|
-
hotkey="p",
|
|
159
|
-
params=(
|
|
160
|
-
ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
|
|
161
|
-
ParamSpec(
|
|
162
|
-
"keep",
|
|
163
|
-
"choice",
|
|
164
|
-
"Keep",
|
|
165
|
-
default="first",
|
|
166
|
-
choices=("first", "last"),
|
|
167
|
-
),
|
|
168
|
-
),
|
|
169
|
-
apply=_apply_dedup_rows,
|
|
170
|
-
),
|
|
171
172
|
Operation(
|
|
172
173
|
key="combine_columns",
|
|
173
174
|
title="Combine columns",
|
|
@@ -212,6 +213,22 @@ OPS: tuple[Operation, ...] = (
|
|
|
212
213
|
),
|
|
213
214
|
apply=_apply_drop_columns,
|
|
214
215
|
),
|
|
216
|
+
Operation(
|
|
217
|
+
key="dedup_rows",
|
|
218
|
+
title="Deduplicate rows",
|
|
219
|
+
hotkey="p",
|
|
220
|
+
params=(
|
|
221
|
+
ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
|
|
222
|
+
ParamSpec(
|
|
223
|
+
"keep",
|
|
224
|
+
"choice",
|
|
225
|
+
"Keep",
|
|
226
|
+
default="first",
|
|
227
|
+
choices=("first", "last"),
|
|
228
|
+
),
|
|
229
|
+
),
|
|
230
|
+
apply=_apply_dedup_rows,
|
|
231
|
+
),
|
|
215
232
|
)
|
|
216
233
|
|
|
217
234
|
OP_REGISTRY: dict[str, Operation] = {op.key: op for op in OPS}
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/screens.py
RENAMED
|
@@ -363,7 +363,11 @@ class OpParamsModal(ModalDialog):
|
|
|
363
363
|
|
|
364
364
|
def compose_body(self) -> ComposeResult:
|
|
365
365
|
self.widgets: dict[str, object] = {}
|
|
366
|
-
candidates =
|
|
366
|
+
candidates = (
|
|
367
|
+
self._date_candidates()
|
|
368
|
+
if self.op.key in ("date_normalize", "date_only")
|
|
369
|
+
else []
|
|
370
|
+
)
|
|
367
371
|
for spec in self.op.params:
|
|
368
372
|
if spec.kind == "column_multi":
|
|
369
373
|
sel = SelectionList(
|
|
@@ -531,8 +535,8 @@ class SaveModal(ModalDialog):
|
|
|
531
535
|
class ScriptNameModal(ModalDialog):
|
|
532
536
|
"""Name the '.ntd' script file when saving a sequence of operations.
|
|
533
537
|
|
|
534
|
-
The
|
|
535
|
-
|
|
538
|
+
The extension is always .ntd: whatever the user types, the saved name
|
|
539
|
+
ends in the fixed script suffix."""
|
|
536
540
|
|
|
537
541
|
dialog_title = "Save script"
|
|
538
542
|
|
|
@@ -547,17 +551,14 @@ class ScriptNameModal(ModalDialog):
|
|
|
547
551
|
id="script_input",
|
|
548
552
|
)
|
|
549
553
|
yield self.path_input
|
|
550
|
-
yield Static(
|
|
551
|
-
f"Default extension: {SCRIPT_SUFFIX} — type another to override",
|
|
552
|
-
classes="help",
|
|
553
|
-
)
|
|
554
|
+
yield Static(f"Scripts always end in {SCRIPT_SUFFIX}", classes="help")
|
|
554
555
|
|
|
555
556
|
def action_ok(self) -> None:
|
|
556
557
|
raw = self.path_input.value.strip()
|
|
557
558
|
if not raw:
|
|
558
559
|
self.app.notify("Type a script file name or path", severity="error")
|
|
559
560
|
return
|
|
560
|
-
target = Path(raw).expanduser()
|
|
561
|
+
target = Path(raw).expanduser().with_suffix(SCRIPT_SUFFIX)
|
|
561
562
|
if target.exists():
|
|
562
563
|
self.app.notify("Script file exists: pick another name", severity="warning")
|
|
563
564
|
return
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/widgets.py
RENAMED
|
@@ -92,13 +92,13 @@ class ColumnSidebar(Widget):
|
|
|
92
92
|
|
|
93
93
|
|
|
94
94
|
def _dtype_mark(dtype: str) -> str:
|
|
95
|
-
"""Compact 3-wide dtype mark for the sidebar: dt (datetime
|
|
96
|
-
str (string), int (integer), dec (float/decimal),
|
|
97
|
-
everything else oth. Each mark is padded to 3 columns
|
|
98
|
-
names left-align."""
|
|
95
|
+
"""Compact 3-wide dtype mark for the sidebar: dt (datetime),
|
|
96
|
+
day (date-only), str (string), int (integer), dec (float/decimal),
|
|
97
|
+
t/f (boolean); everything else oth. Each mark is padded to 3 columns
|
|
98
|
+
so column names left-align."""
|
|
99
99
|
for prefix, mark in (
|
|
100
100
|
("Datetime", "dt "),
|
|
101
|
-
("Date", "
|
|
101
|
+
("Date", "day"),
|
|
102
102
|
("Time", "dt "),
|
|
103
103
|
("String", "str"),
|
|
104
104
|
("Categorical", "str"),
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.4}/src/normalize_tabular_data/__main__.py
RENAMED
|
File without changes
|