normalize-tabular-data 0.1.4__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/PKG-INFO +14 -6
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/README.md +12 -5
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/pyproject.toml +2 -1
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/pyproject.toml.orig +2 -1
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/__init__.py +4 -3
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/app.py +2 -1
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/io.py +47 -4
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/ops.py +1 -1
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/LICENSE +0 -0
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/__main__.py +0 -0
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/screens.py +0 -0
- {normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/widgets.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: normalize-tabular-data
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: TUI for normalizing tabular data with Polars
|
|
5
5
|
Author: David Mertz, Ph.D.
|
|
6
6
|
Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
|
|
@@ -12,6 +12,7 @@ Requires-Dist: polars>=1.44,<2
|
|
|
12
12
|
Requires-Dist: fastexcel>=0.10
|
|
13
13
|
Requires-Dist: xlsxwriter>=0.9
|
|
14
14
|
Requires-Dist: platformdirs>=3.0
|
|
15
|
+
Requires-Dist: chardet>=7.0.0
|
|
15
16
|
Requires-Python: >=3.14
|
|
16
17
|
Description-Content-Type: text/markdown
|
|
17
18
|
|
|
@@ -20,10 +21,11 @@ Description-Content-Type: text/markdown
|
|
|
20
21
|
A terminal UI (Textual) for interactively normalizing tabular data, backed by
|
|
21
22
|
[polars](https://pola.rs) for speed and
|
|
22
23
|
[gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
|
|
23
|
-
date parsing at Rust speed.
|
|
24
|
+
date parsing at Rust speed. NTD also supports purely command-line operations
|
|
25
|
+
for use in scripting.
|
|
24
26
|
|
|
25
|
-
|
|
26
|
-
|
|
27
|
+
**Quick start**: `uvx normalize-tabular-data`. No installation required if
|
|
28
|
+
you have `uv`.
|
|
27
29
|
|
|
28
30
|
Load a file, see a preview, build up a pipeline of normalization operations
|
|
29
31
|
(normalize messy dates, trim whitespace, rename a column, deduplicate,
|
|
@@ -34,6 +36,11 @@ combine/split columns, drop columns), then save the cleaned result.
|
|
|
34
36
|
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
35
37
|
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
36
38
|
|
|
39
|
+
Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
|
|
40
|
+
Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
|
|
41
|
+
[chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
|
|
42
|
+
before parsing.
|
|
43
|
+
|
|
37
44
|
Large files are previewed with a random sample of 250 rows; every operation
|
|
38
45
|
still runs on the full data.
|
|
39
46
|
|
|
@@ -44,7 +51,8 @@ still runs on the full data.
|
|
|
44
51
|
Installation and launch options — persistent install, ephemeral `uvx`
|
|
45
52
|
runs, running from a git checkout, the single-file launchers for
|
|
46
53
|
Linux/macOS/Windows, and installing them as desktop icons — are
|
|
47
|
-
described in
|
|
54
|
+
described in
|
|
55
|
+
[docs/running-the-tool.md](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/docs/running-the-tool.md).
|
|
48
56
|
|
|
49
57
|
## Capabilities in the TUI
|
|
50
58
|
|
|
@@ -181,4 +189,4 @@ credential aborts the upload instead of prompting mid-run.
|
|
|
181
189
|
|
|
182
190
|
BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
|
|
183
191
|
|
|
184
|
-
See [LICENSE](LICENSE).
|
|
192
|
+
See [LICENSE](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/LICENSE).
|
|
@@ -3,10 +3,11 @@
|
|
|
3
3
|
A terminal UI (Textual) for interactively normalizing tabular data, backed by
|
|
4
4
|
[polars](https://pola.rs) for speed and
|
|
5
5
|
[gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
|
|
6
|
-
date parsing at Rust speed.
|
|
6
|
+
date parsing at Rust speed. NTD also supports purely command-line operations
|
|
7
|
+
for use in scripting.
|
|
7
8
|
|
|
8
|
-
|
|
9
|
-
|
|
9
|
+
**Quick start**: `uvx normalize-tabular-data`. No installation required if
|
|
10
|
+
you have `uv`.
|
|
10
11
|
|
|
11
12
|
Load a file, see a preview, build up a pipeline of normalization operations
|
|
12
13
|
(normalize messy dates, trim whitespace, rename a column, deduplicate,
|
|
@@ -17,6 +18,11 @@ combine/split columns, drop columns), then save the cleaned result.
|
|
|
17
18
|
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
18
19
|
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
19
20
|
|
|
21
|
+
Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
|
|
22
|
+
Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
|
|
23
|
+
[chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
|
|
24
|
+
before parsing.
|
|
25
|
+
|
|
20
26
|
Large files are previewed with a random sample of 250 rows; every operation
|
|
21
27
|
still runs on the full data.
|
|
22
28
|
|
|
@@ -27,7 +33,8 @@ still runs on the full data.
|
|
|
27
33
|
Installation and launch options — persistent install, ephemeral `uvx`
|
|
28
34
|
runs, running from a git checkout, the single-file launchers for
|
|
29
35
|
Linux/macOS/Windows, and installing them as desktop icons — are
|
|
30
|
-
described in
|
|
36
|
+
described in
|
|
37
|
+
[docs/running-the-tool.md](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/docs/running-the-tool.md).
|
|
31
38
|
|
|
32
39
|
## Capabilities in the TUI
|
|
33
40
|
|
|
@@ -164,4 +171,4 @@ credential aborts the upload instead of prompting mid-run.
|
|
|
164
171
|
|
|
165
172
|
BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
|
|
166
173
|
|
|
167
|
-
See [LICENSE](LICENSE).
|
|
174
|
+
See [LICENSE](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/LICENSE).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
@@ -13,6 +13,7 @@ dependencies = [
|
|
|
13
13
|
"fastexcel>=0.10",
|
|
14
14
|
"xlsxwriter>=0.9",
|
|
15
15
|
"platformdirs>=3.0",
|
|
16
|
+
"chardet>=7.0.0",
|
|
16
17
|
]
|
|
17
18
|
|
|
18
19
|
[[project.authors]]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
@@ -16,6 +16,7 @@ dependencies = [
|
|
|
16
16
|
"fastexcel>=0.10",
|
|
17
17
|
"xlsxwriter>=0.9",
|
|
18
18
|
"platformdirs>=3.0",
|
|
19
|
+
"chardet>=7.0.0",
|
|
19
20
|
]
|
|
20
21
|
|
|
21
22
|
[dependency-groups]
|
{normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/__init__.py
RENAMED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""normalize-tabular-data: TUI for normalizing tabular data with
|
|
1
|
+
"""normalize-tabular-data: TUI for normalizing tabular data with Polars."""
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
@@ -6,7 +6,7 @@ import argparse
|
|
|
6
6
|
import sys
|
|
7
7
|
from pathlib import Path
|
|
8
8
|
|
|
9
|
-
__version__ = "0.
|
|
9
|
+
__version__ = "0.2.0"
|
|
10
10
|
|
|
11
11
|
|
|
12
12
|
def _run_script(path: Path, script_path: Path, output: Path | None) -> int:
|
|
@@ -145,7 +145,8 @@ def main() -> None:
|
|
|
145
145
|
except ImportError as exc: # helpful message if an optional engine is missing
|
|
146
146
|
print(
|
|
147
147
|
f"normalize-tabular-data is missing a dependency ({exc}).\n"
|
|
148
|
-
"Reinstall with:
|
|
148
|
+
"Reinstall with: "
|
|
149
|
+
"uv tool install --force --reinstall normalize-tabular-data",
|
|
149
150
|
file=sys.stderr,
|
|
150
151
|
)
|
|
151
152
|
raise SystemExit(1)
|
{normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/app.py
RENAMED
|
@@ -21,6 +21,7 @@ from textual.screen import Screen
|
|
|
21
21
|
from textual.widgets import DataTable, Header, Label, ListItem, ListView
|
|
22
22
|
|
|
23
23
|
from normalize_tabular_data import io
|
|
24
|
+
from normalize_tabular_data import __version__
|
|
24
25
|
from normalize_tabular_data.io import SCRIPT_SUFFIX
|
|
25
26
|
from normalize_tabular_data.ops import (
|
|
26
27
|
OP_REGISTRY,
|
|
@@ -207,7 +208,7 @@ class AppHeader(Header):
|
|
|
207
208
|
|
|
208
209
|
def compose(self) -> ComposeResult:
|
|
209
210
|
yield from super().compose()
|
|
210
|
-
yield Label("normalize-tabular-data", classes="app_name")
|
|
211
|
+
yield Label(f"normalize-tabular-data {__version__}", classes="app_name")
|
|
211
212
|
|
|
212
213
|
|
|
213
214
|
class MainScreen(Screen[None]):
|
{normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/io.py
RENAMED
|
@@ -1,9 +1,11 @@
|
|
|
1
|
-
"""Format dispatch: suffix ->
|
|
1
|
+
"""Format dispatch: suffix -> Polars read/write. No Textual imports."""
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
import chardet
|
|
5
6
|
import json
|
|
6
7
|
import re
|
|
8
|
+
from io import BytesIO
|
|
7
9
|
from pathlib import Path
|
|
8
10
|
|
|
9
11
|
import polars as pl
|
|
@@ -41,17 +43,55 @@ def detect_format(path: Path) -> str:
|
|
|
41
43
|
return fmt
|
|
42
44
|
|
|
43
45
|
|
|
46
|
+
# bytes probed for encoding detection; large files stay fast
|
|
47
|
+
_SNIFF_BYTES = 100_000
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _decoded_text(data: bytes) -> bytes:
|
|
51
|
+
"""`data`, transcoded to UTF-8 (polars parses text formats as UTF-8 only).
|
|
52
|
+
|
|
53
|
+
Already-valid UTF-8 (the common case) passes through untouched. Anything
|
|
54
|
+
else is decoded with the encoding chardet sniffs from the leading bytes
|
|
55
|
+
-- ISO 8859-3, Windows-1256, Shift_JIS-2004, UTF-16, etc. -- and
|
|
56
|
+
re-encoded. When sniffer or decodng fails, an errors=replace decode
|
|
57
|
+
keeps the load alive instead of erroring out."""
|
|
58
|
+
if not data:
|
|
59
|
+
return data
|
|
60
|
+
if data.startswith(b"\xef\xbb\xbf"): # UTF-8 BOM
|
|
61
|
+
data = data[3:]
|
|
62
|
+
try:
|
|
63
|
+
data.decode("utf-8")
|
|
64
|
+
return data
|
|
65
|
+
except UnicodeDecodeError:
|
|
66
|
+
pass
|
|
67
|
+
encoding = chardet.detect(data[:_SNIFF_BYTES])["encoding"]
|
|
68
|
+
try:
|
|
69
|
+
decoded = data.decode(encoding)
|
|
70
|
+
except (LookupError, UnicodeDecodeError):
|
|
71
|
+
decoded = data.decode("utf-8", errors="replace")
|
|
72
|
+
return decoded.removeprefix("").encode()
|
|
73
|
+
|
|
74
|
+
|
|
44
75
|
def _read_tabular(path: Path, separator: str = ",") -> pl.DataFrame:
|
|
45
76
|
"""CSV/TSV read with a retry fallback.
|
|
46
77
|
|
|
47
78
|
Inference looks at up to 10,000 rows; dtype-incompatible values can
|
|
48
79
|
still appear later in a large file and fail parsing. When that happens,
|
|
49
80
|
re-read with zero-length inference so every column stays String instead
|
|
50
|
-
of erroring out.
|
|
81
|
+
of erroring out. The same retry also covers a non-UTF-8 encoding: the
|
|
82
|
+
sniffer (`_decoded_text`) transcoded the bytes to UTF-8."""
|
|
51
83
|
try:
|
|
52
84
|
return pl.read_csv(path, separator=separator, infer_schema_length=10_000)
|
|
53
85
|
except pl.exceptions.PolarsError:
|
|
54
|
-
|
|
86
|
+
data = _decoded_text(path.read_bytes())
|
|
87
|
+
try:
|
|
88
|
+
return pl.read_csv(
|
|
89
|
+
BytesIO(data), separator=separator, infer_schema_length=10_000
|
|
90
|
+
)
|
|
91
|
+
except pl.exceptions.PolarsError:
|
|
92
|
+
return pl.read_csv(
|
|
93
|
+
BytesIO(data), separator=separator, infer_schema_length=0
|
|
94
|
+
)
|
|
55
95
|
|
|
56
96
|
|
|
57
97
|
def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
|
|
@@ -61,7 +101,10 @@ def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
|
|
|
61
101
|
if fmt == "tsv":
|
|
62
102
|
return _read_tabular(path, separator="\t")
|
|
63
103
|
if fmt == "jsonl":
|
|
64
|
-
|
|
104
|
+
try:
|
|
105
|
+
return pl.read_ndjson(path)
|
|
106
|
+
except pl.exceptions.PolarsError:
|
|
107
|
+
return pl.read_ndjson(BytesIO(_decoded_text(path.read_bytes())))
|
|
65
108
|
if fmt == "parquet":
|
|
66
109
|
return pl.read_parquet(path)
|
|
67
110
|
if fmt == "xlsx":
|
|
File without changes
|
{normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/__main__.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/screens.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.4 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/widgets.py
RENAMED
|
File without changes
|