normalize-tabular-data 0.1.4__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: normalize-tabular-data
3
- Version: 0.1.4
3
+ Version: 0.2.0
4
4
  Summary: TUI for normalizing tabular data with Polars
5
5
  Author: David Mertz, Ph.D.
6
6
  Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
@@ -12,6 +12,7 @@ Requires-Dist: polars>=1.44,<2
12
12
  Requires-Dist: fastexcel>=0.10
13
13
  Requires-Dist: xlsxwriter>=0.9
14
14
  Requires-Dist: platformdirs>=3.0
15
+ Requires-Dist: chardet>=7.0.0
15
16
  Requires-Python: >=3.14
16
17
  Description-Content-Type: text/markdown
17
18
 
@@ -20,10 +21,11 @@ Description-Content-Type: text/markdown
20
21
  A terminal UI (Textual) for interactively normalizing tabular data, backed by
21
22
  [polars](https://pola.rs) for speed and
22
23
  [gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
23
- date parsing at Rust speed.
24
+ date parsing at Rust speed. NTD also supports purely command-line operations
25
+ for use in scripting.
24
26
 
25
- NTD also supports purely command-line operations for use in scripting, launched
26
- with `uvx` with no other requirements that are not dynamically downloaded.
27
+ **Quick start**: `uvx normalize-tabular-data`. No installation required if
28
+ you have `uv`.
27
29
 
28
30
  Load a file, see a preview, build up a pipeline of normalization operations
29
31
  (normalize messy dates, trim whitespace, rename a column, deduplicate,
@@ -34,6 +36,11 @@ combine/split columns, drop columns), then save the cleaned result.
34
36
  Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
35
37
  Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
36
38
 
39
+ Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
40
+ Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
41
+ [chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
42
+ before parsing.
43
+
37
44
  Large files are previewed with a random sample of 250 rows; every operation
38
45
  still runs on the full data.
39
46
 
@@ -44,7 +51,8 @@ still runs on the full data.
44
51
  Installation and launch options — persistent install, ephemeral `uvx`
45
52
  runs, running from a git checkout, the single-file launchers for
46
53
  Linux/macOS/Windows, and installing them as desktop icons — are
47
- described in [docs/running-the-tool.md](docs/running-the-tool.md).
54
+ described in
55
+ [docs/running-the-tool.md](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/docs/running-the-tool.md).
48
56
 
49
57
  ## Capabilities in the TUI
50
58
 
@@ -181,4 +189,4 @@ credential aborts the upload instead of prompting mid-run.
181
189
 
182
190
  BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
183
191
 
184
- See [LICENSE](LICENSE).
192
+ See [LICENSE](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/LICENSE).
@@ -3,10 +3,11 @@
3
3
  A terminal UI (Textual) for interactively normalizing tabular data, backed by
4
4
  [polars](https://pola.rs) for speed and
5
5
  [gnosis-date-parser](https://pypi.org/project/gnosis-date-parser/) for fuzzy
6
- date parsing at Rust speed.
6
+ date parsing at Rust speed. NTD also supports purely command-line operations
7
+ for use in scripting.
7
8
 
8
- NTD also supports purely command-line operations for use in scripting, launched
9
- with `uvx` with no other requirements that are not dynamically downloaded.
9
+ **Quick start**: `uvx normalize-tabular-data`. No installation required if
10
+ you have `uv`.
10
11
 
11
12
  Load a file, see a preview, build up a pipeline of normalization operations
12
13
  (normalize messy dates, trim whitespace, rename a column, deduplicate,
@@ -17,6 +18,11 @@ combine/split columns, drop columns), then save the cleaned result.
17
18
  Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
18
19
  Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
19
20
 
21
+ Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
22
+ Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
23
+ [chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
24
+ before parsing.
25
+
20
26
  Large files are previewed with a random sample of 250 rows; every operation
21
27
  still runs on the full data.
22
28
 
@@ -27,7 +33,8 @@ still runs on the full data.
27
33
  Installation and launch options — persistent install, ephemeral `uvx`
28
34
  runs, running from a git checkout, the single-file launchers for
29
35
  Linux/macOS/Windows, and installing them as desktop icons — are
30
- described in [docs/running-the-tool.md](docs/running-the-tool.md).
36
+ described in
37
+ [docs/running-the-tool.md](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/docs/running-the-tool.md).
31
38
 
32
39
  ## Capabilities in the TUI
33
40
 
@@ -164,4 +171,4 @@ credential aborts the upload instead of prompting mid-run.
164
171
 
165
172
  BSD 2-Clause. Copyright (c) 2026, Service Employees International Union (SEIU).
166
173
 
167
- See [LICENSE](LICENSE).
174
+ See [LICENSE](https://github.com/SEIU-Tech/normalize-tabular-data/blob/main/LICENSE).
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "normalize-tabular-data"
3
- version = "0.1.4"
3
+ version = "0.2.0"
4
4
  description = "TUI for normalizing tabular data with Polars"
5
5
  readme = "README.md"
6
6
  license = "BSD-2-Clause"
@@ -13,6 +13,7 @@ dependencies = [
13
13
  "fastexcel>=0.10",
14
14
  "xlsxwriter>=0.9",
15
15
  "platformdirs>=3.0",
16
+ "chardet>=7.0.0",
16
17
  ]
17
18
 
18
19
  [[project.authors]]
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "normalize-tabular-data"
3
- version = "0.1.4"
3
+ version = "0.2.0"
4
4
  description = "TUI for normalizing tabular data with Polars"
5
5
  readme = "README.md"
6
6
  license = "BSD-2-Clause"
@@ -16,6 +16,7 @@ dependencies = [
16
16
  "fastexcel>=0.10",
17
17
  "xlsxwriter>=0.9",
18
18
  "platformdirs>=3.0",
19
+ "chardet>=7.0.0",
19
20
  ]
20
21
 
21
22
  [dependency-groups]
@@ -1,4 +1,4 @@
1
- """normalize-tabular-data: TUI for normalizing tabular data with polars."""
1
+ """normalize-tabular-data: TUI for normalizing tabular data with Polars."""
2
2
 
3
3
  from __future__ import annotations
4
4
 
@@ -6,7 +6,7 @@ import argparse
6
6
  import sys
7
7
  from pathlib import Path
8
8
 
9
- __version__ = "0.1.4"
9
+ __version__ = "0.2.0"
10
10
 
11
11
 
12
12
  def _run_script(path: Path, script_path: Path, output: Path | None) -> int:
@@ -145,7 +145,8 @@ def main() -> None:
145
145
  except ImportError as exc: # helpful message if an optional engine is missing
146
146
  print(
147
147
  f"normalize-tabular-data is missing a dependency ({exc}).\n"
148
- "Reinstall with: uv tool install --force --reinstall normalize-tabular-data",
148
+ "Reinstall with: "
149
+ "uv tool install --force --reinstall normalize-tabular-data",
149
150
  file=sys.stderr,
150
151
  )
151
152
  raise SystemExit(1)
@@ -21,6 +21,7 @@ from textual.screen import Screen
21
21
  from textual.widgets import DataTable, Header, Label, ListItem, ListView
22
22
 
23
23
  from normalize_tabular_data import io
24
+ from normalize_tabular_data import __version__
24
25
  from normalize_tabular_data.io import SCRIPT_SUFFIX
25
26
  from normalize_tabular_data.ops import (
26
27
  OP_REGISTRY,
@@ -207,7 +208,7 @@ class AppHeader(Header):
207
208
 
208
209
  def compose(self) -> ComposeResult:
209
210
  yield from super().compose()
210
- yield Label("normalize-tabular-data", classes="app_name")
211
+ yield Label(f"normalize-tabular-data {__version__}", classes="app_name")
211
212
 
212
213
 
213
214
  class MainScreen(Screen[None]):
@@ -1,9 +1,11 @@
1
- """Format dispatch: suffix -> polars read/write. No Textual imports."""
1
+ """Format dispatch: suffix -> Polars read/write. No Textual imports."""
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ import chardet
5
6
  import json
6
7
  import re
8
+ from io import BytesIO
7
9
  from pathlib import Path
8
10
 
9
11
  import polars as pl
@@ -41,17 +43,55 @@ def detect_format(path: Path) -> str:
41
43
  return fmt
42
44
 
43
45
 
46
+ # bytes probed for encoding detection; large files stay fast
47
+ _SNIFF_BYTES = 100_000
48
+
49
+
50
+ def _decoded_text(data: bytes) -> bytes:
51
+ """`data`, transcoded to UTF-8 (polars parses text formats as UTF-8 only).
52
+
53
+ Already-valid UTF-8 (the common case) passes through untouched. Anything
54
+ else is decoded with the encoding chardet sniffs from the leading bytes
55
+ -- ISO 8859-3, Windows-1256, Shift_JIS-2004, UTF-16, etc. -- and
56
+ re-encoded. When sniffer or decodng fails, an errors=replace decode
57
+ keeps the load alive instead of erroring out."""
58
+ if not data:
59
+ return data
60
+ if data.startswith(b"\xef\xbb\xbf"): # UTF-8 BOM
61
+ data = data[3:]
62
+ try:
63
+ data.decode("utf-8")
64
+ return data
65
+ except UnicodeDecodeError:
66
+ pass
67
+ encoding = chardet.detect(data[:_SNIFF_BYTES])["encoding"]
68
+ try:
69
+ decoded = data.decode(encoding)
70
+ except (LookupError, UnicodeDecodeError):
71
+ decoded = data.decode("utf-8", errors="replace")
72
+ return decoded.removeprefix("").encode()
73
+
74
+
44
75
  def _read_tabular(path: Path, separator: str = ",") -> pl.DataFrame:
45
76
  """CSV/TSV read with a retry fallback.
46
77
 
47
78
  Inference looks at up to 10,000 rows; dtype-incompatible values can
48
79
  still appear later in a large file and fail parsing. When that happens,
49
80
  re-read with zero-length inference so every column stays String instead
50
- of erroring out."""
81
+ of erroring out. The same retry also covers a non-UTF-8 encoding: the
82
+ sniffer (`_decoded_text`) transcoded the bytes to UTF-8."""
51
83
  try:
52
84
  return pl.read_csv(path, separator=separator, infer_schema_length=10_000)
53
85
  except pl.exceptions.PolarsError:
54
- return pl.read_csv(path, separator=separator, infer_schema_length=0)
86
+ data = _decoded_text(path.read_bytes())
87
+ try:
88
+ return pl.read_csv(
89
+ BytesIO(data), separator=separator, infer_schema_length=10_000
90
+ )
91
+ except pl.exceptions.PolarsError:
92
+ return pl.read_csv(
93
+ BytesIO(data), separator=separator, infer_schema_length=0
94
+ )
55
95
 
56
96
 
57
97
  def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
@@ -61,7 +101,10 @@ def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
61
101
  if fmt == "tsv":
62
102
  return _read_tabular(path, separator="\t")
63
103
  if fmt == "jsonl":
64
- return pl.read_ndjson(path)
104
+ try:
105
+ return pl.read_ndjson(path)
106
+ except pl.exceptions.PolarsError:
107
+ return pl.read_ndjson(BytesIO(_decoded_text(path.read_bytes())))
65
108
  if fmt == "parquet":
66
109
  return pl.read_parquet(path)
67
110
  if fmt == "xlsx":
@@ -1,6 +1,6 @@
1
1
  """Normalization engine: operation registry and re-computing pipeline.
2
2
 
3
- Pure polars — no Textual imports here, so the whole engine is testable
3
+ Pure Polars — no Textual imports here, so the whole engine is testable
4
4
  without a terminal.
5
5
  """
6
6