normalize-tabular-data 0.1.5__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/PKG-INFO +7 -1
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/README.md +5 -0
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/pyproject.toml +2 -1
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/pyproject.toml.orig +2 -1
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/__init__.py +1 -1
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/app.py +2 -1
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/io.py +46 -3
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/LICENSE +0 -0
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/__main__.py +0 -0
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/ops.py +0 -0
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/screens.py +0 -0
- {normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/widgets.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: normalize-tabular-data
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: TUI for normalizing tabular data with Polars
|
|
5
5
|
Author: David Mertz, Ph.D.
|
|
6
6
|
Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
|
|
@@ -12,6 +12,7 @@ Requires-Dist: polars>=1.44,<2
|
|
|
12
12
|
Requires-Dist: fastexcel>=0.10
|
|
13
13
|
Requires-Dist: xlsxwriter>=0.9
|
|
14
14
|
Requires-Dist: platformdirs>=3.0
|
|
15
|
+
Requires-Dist: chardet>=7.0.0
|
|
15
16
|
Requires-Python: >=3.14
|
|
16
17
|
Description-Content-Type: text/markdown
|
|
17
18
|
|
|
@@ -35,6 +36,11 @@ combine/split columns, drop columns), then save the cleaned result.
|
|
|
35
36
|
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
36
37
|
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
37
38
|
|
|
39
|
+
Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
|
|
40
|
+
Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
|
|
41
|
+
[chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
|
|
42
|
+
before parsing.
|
|
43
|
+
|
|
38
44
|
Large files are previewed with a random sample of 250 rows; every operation
|
|
39
45
|
still runs on the full data.
|
|
40
46
|
|
|
@@ -18,6 +18,11 @@ combine/split columns, drop columns), then save the cleaned result.
|
|
|
18
18
|
Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
|
|
19
19
|
Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
|
|
20
20
|
|
|
21
|
+
Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
|
|
22
|
+
Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
|
|
23
|
+
[chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
|
|
24
|
+
before parsing.
|
|
25
|
+
|
|
21
26
|
Large files are previewed with a random sample of 250 rows; every operation
|
|
22
27
|
still runs on the full data.
|
|
23
28
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
@@ -13,6 +13,7 @@ dependencies = [
|
|
|
13
13
|
"fastexcel>=0.10",
|
|
14
14
|
"xlsxwriter>=0.9",
|
|
15
15
|
"platformdirs>=3.0",
|
|
16
|
+
"chardet>=7.0.0",
|
|
16
17
|
]
|
|
17
18
|
|
|
18
19
|
[[project.authors]]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
@@ -16,6 +16,7 @@ dependencies = [
|
|
|
16
16
|
"fastexcel>=0.10",
|
|
17
17
|
"xlsxwriter>=0.9",
|
|
18
18
|
"platformdirs>=3.0",
|
|
19
|
+
"chardet>=7.0.0",
|
|
19
20
|
]
|
|
20
21
|
|
|
21
22
|
[dependency-groups]
|
{normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/app.py
RENAMED
|
@@ -21,6 +21,7 @@ from textual.screen import Screen
|
|
|
21
21
|
from textual.widgets import DataTable, Header, Label, ListItem, ListView
|
|
22
22
|
|
|
23
23
|
from normalize_tabular_data import io
|
|
24
|
+
from normalize_tabular_data import __version__
|
|
24
25
|
from normalize_tabular_data.io import SCRIPT_SUFFIX
|
|
25
26
|
from normalize_tabular_data.ops import (
|
|
26
27
|
OP_REGISTRY,
|
|
@@ -207,7 +208,7 @@ class AppHeader(Header):
|
|
|
207
208
|
|
|
208
209
|
def compose(self) -> ComposeResult:
|
|
209
210
|
yield from super().compose()
|
|
210
|
-
yield Label("normalize-tabular-data", classes="app_name")
|
|
211
|
+
yield Label(f"normalize-tabular-data {__version__}", classes="app_name")
|
|
211
212
|
|
|
212
213
|
|
|
213
214
|
class MainScreen(Screen[None]):
|
{normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/io.py
RENAMED
|
@@ -2,8 +2,10 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
import chardet
|
|
5
6
|
import json
|
|
6
7
|
import re
|
|
8
|
+
from io import BytesIO
|
|
7
9
|
from pathlib import Path
|
|
8
10
|
|
|
9
11
|
import polars as pl
|
|
@@ -41,17 +43,55 @@ def detect_format(path: Path) -> str:
|
|
|
41
43
|
return fmt
|
|
42
44
|
|
|
43
45
|
|
|
46
|
+
# bytes probed for encoding detection; large files stay fast
|
|
47
|
+
_SNIFF_BYTES = 100_000
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _decoded_text(data: bytes) -> bytes:
|
|
51
|
+
"""`data`, transcoded to UTF-8 (polars parses text formats as UTF-8 only).
|
|
52
|
+
|
|
53
|
+
Already-valid UTF-8 (the common case) passes through untouched. Anything
|
|
54
|
+
else is decoded with the encoding chardet sniffs from the leading bytes
|
|
55
|
+
-- ISO 8859-3, Windows-1256, Shift_JIS-2004, UTF-16, etc. -- and
|
|
56
|
+
re-encoded. When sniffer or decodng fails, an errors=replace decode
|
|
57
|
+
keeps the load alive instead of erroring out."""
|
|
58
|
+
if not data:
|
|
59
|
+
return data
|
|
60
|
+
if data.startswith(b"\xef\xbb\xbf"): # UTF-8 BOM
|
|
61
|
+
data = data[3:]
|
|
62
|
+
try:
|
|
63
|
+
data.decode("utf-8")
|
|
64
|
+
return data
|
|
65
|
+
except UnicodeDecodeError:
|
|
66
|
+
pass
|
|
67
|
+
encoding = chardet.detect(data[:_SNIFF_BYTES])["encoding"]
|
|
68
|
+
try:
|
|
69
|
+
decoded = data.decode(encoding)
|
|
70
|
+
except (LookupError, UnicodeDecodeError):
|
|
71
|
+
decoded = data.decode("utf-8", errors="replace")
|
|
72
|
+
return decoded.removeprefix("").encode()
|
|
73
|
+
|
|
74
|
+
|
|
44
75
|
def _read_tabular(path: Path, separator: str = ",") -> pl.DataFrame:
|
|
45
76
|
"""CSV/TSV read with a retry fallback.
|
|
46
77
|
|
|
47
78
|
Inference looks at up to 10,000 rows; dtype-incompatible values can
|
|
48
79
|
still appear later in a large file and fail parsing. When that happens,
|
|
49
80
|
re-read with zero-length inference so every column stays String instead
|
|
50
|
-
of erroring out.
|
|
81
|
+
of erroring out. The same retry also covers a non-UTF-8 encoding: the
|
|
82
|
+
sniffer (`_decoded_text`) transcoded the bytes to UTF-8."""
|
|
51
83
|
try:
|
|
52
84
|
return pl.read_csv(path, separator=separator, infer_schema_length=10_000)
|
|
53
85
|
except pl.exceptions.PolarsError:
|
|
54
|
-
|
|
86
|
+
data = _decoded_text(path.read_bytes())
|
|
87
|
+
try:
|
|
88
|
+
return pl.read_csv(
|
|
89
|
+
BytesIO(data), separator=separator, infer_schema_length=10_000
|
|
90
|
+
)
|
|
91
|
+
except pl.exceptions.PolarsError:
|
|
92
|
+
return pl.read_csv(
|
|
93
|
+
BytesIO(data), separator=separator, infer_schema_length=0
|
|
94
|
+
)
|
|
55
95
|
|
|
56
96
|
|
|
57
97
|
def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
|
|
@@ -61,7 +101,10 @@ def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
|
|
|
61
101
|
if fmt == "tsv":
|
|
62
102
|
return _read_tabular(path, separator="\t")
|
|
63
103
|
if fmt == "jsonl":
|
|
64
|
-
|
|
104
|
+
try:
|
|
105
|
+
return pl.read_ndjson(path)
|
|
106
|
+
except pl.exceptions.PolarsError:
|
|
107
|
+
return pl.read_ndjson(BytesIO(_decoded_text(path.read_bytes())))
|
|
65
108
|
if fmt == "parquet":
|
|
66
109
|
return pl.read_parquet(path)
|
|
67
110
|
if fmt == "xlsx":
|
|
File without changes
|
{normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/__main__.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/ops.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/screens.py
RENAMED
|
File without changes
|
{normalize_tabular_data-0.1.5 → normalize_tabular_data-0.2.0}/src/normalize_tabular_data/widgets.py
RENAMED
|
File without changes
|