normalize-tabular-data 0.1.5__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: normalize-tabular-data
3
- Version: 0.1.5
3
+ Version: 0.2.0
4
4
  Summary: TUI for normalizing tabular data with Polars
5
5
  Author: David Mertz, Ph.D.
6
6
  Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
@@ -12,6 +12,7 @@ Requires-Dist: polars>=1.44,<2
12
12
  Requires-Dist: fastexcel>=0.10
13
13
  Requires-Dist: xlsxwriter>=0.9
14
14
  Requires-Dist: platformdirs>=3.0
15
+ Requires-Dist: chardet>=7.0.0
15
16
  Requires-Python: >=3.14
16
17
  Description-Content-Type: text/markdown
17
18
 
@@ -35,6 +36,11 @@ combine/split columns, drop columns), then save the cleaned result.
35
36
  Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
36
37
  Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
37
38
 
39
+ Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
40
+ Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
41
+ [chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
42
+ before parsing.
43
+
38
44
  Large files are previewed with a random sample of 250 rows; every operation
39
45
  still runs on the full data.
40
46
 
@@ -18,6 +18,11 @@ combine/split columns, drop columns), then save the cleaned result.
18
18
  Reads: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`/`.xls` — first sheet).
19
19
  Writes: CSV, TSV, JSON lines, Parquet, Excel (`.xlsx`).
20
20
 
21
+ Text formats are parsed as UTF-8; a file that is not (ISO 8859-3,
22
+ Windows-1256, Shift_JIS-2004, UTF-16, ...) has its encoding sniffed with
23
+ [chardet](https://pypi.org/project/chardet/) and is transcoded to UTF-8
24
+ before parsing.
25
+
21
26
  Large files are previewed with a random sample of 250 rows; every operation
22
27
  still runs on the full data.
23
28
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "normalize-tabular-data"
3
- version = "0.1.5"
3
+ version = "0.2.0"
4
4
  description = "TUI for normalizing tabular data with Polars"
5
5
  readme = "README.md"
6
6
  license = "BSD-2-Clause"
@@ -13,6 +13,7 @@ dependencies = [
13
13
  "fastexcel>=0.10",
14
14
  "xlsxwriter>=0.9",
15
15
  "platformdirs>=3.0",
16
+ "chardet>=7.0.0",
16
17
  ]
17
18
 
18
19
  [[project.authors]]
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "normalize-tabular-data"
3
- version = "0.1.5"
3
+ version = "0.2.0"
4
4
  description = "TUI for normalizing tabular data with Polars"
5
5
  readme = "README.md"
6
6
  license = "BSD-2-Clause"
@@ -16,6 +16,7 @@ dependencies = [
16
16
  "fastexcel>=0.10",
17
17
  "xlsxwriter>=0.9",
18
18
  "platformdirs>=3.0",
19
+ "chardet>=7.0.0",
19
20
  ]
20
21
 
21
22
  [dependency-groups]
@@ -6,7 +6,7 @@ import argparse
6
6
  import sys
7
7
  from pathlib import Path
8
8
 
9
- __version__ = "0.1.5"
9
+ __version__ = "0.2.0"
10
10
 
11
11
 
12
12
  def _run_script(path: Path, script_path: Path, output: Path | None) -> int:
@@ -21,6 +21,7 @@ from textual.screen import Screen
21
21
  from textual.widgets import DataTable, Header, Label, ListItem, ListView
22
22
 
23
23
  from normalize_tabular_data import io
24
+ from normalize_tabular_data import __version__
24
25
  from normalize_tabular_data.io import SCRIPT_SUFFIX
25
26
  from normalize_tabular_data.ops import (
26
27
  OP_REGISTRY,
@@ -207,7 +208,7 @@ class AppHeader(Header):
207
208
 
208
209
  def compose(self) -> ComposeResult:
209
210
  yield from super().compose()
210
- yield Label("normalize-tabular-data", classes="app_name")
211
+ yield Label(f"normalize-tabular-data {__version__}", classes="app_name")
211
212
 
212
213
 
213
214
  class MainScreen(Screen[None]):
@@ -2,8 +2,10 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ import chardet
5
6
  import json
6
7
  import re
8
+ from io import BytesIO
7
9
  from pathlib import Path
8
10
 
9
11
  import polars as pl
@@ -41,17 +43,55 @@ def detect_format(path: Path) -> str:
41
43
  return fmt
42
44
 
43
45
 
46
+ # bytes probed for encoding detection; large files stay fast
47
+ _SNIFF_BYTES = 100_000
48
+
49
+
50
+ def _decoded_text(data: bytes) -> bytes:
51
+ """`data`, transcoded to UTF-8 (polars parses text formats as UTF-8 only).
52
+
53
+ Already-valid UTF-8 (the common case) passes through untouched. Anything
54
+ else is decoded with the encoding chardet sniffs from the leading bytes
55
+ -- ISO 8859-3, Windows-1256, Shift_JIS-2004, UTF-16, etc. -- and
56
+ re-encoded. When sniffer or decodng fails, an errors=replace decode
57
+ keeps the load alive instead of erroring out."""
58
+ if not data:
59
+ return data
60
+ if data.startswith(b"\xef\xbb\xbf"): # UTF-8 BOM
61
+ data = data[3:]
62
+ try:
63
+ data.decode("utf-8")
64
+ return data
65
+ except UnicodeDecodeError:
66
+ pass
67
+ encoding = chardet.detect(data[:_SNIFF_BYTES])["encoding"]
68
+ try:
69
+ decoded = data.decode(encoding)
70
+ except (LookupError, UnicodeDecodeError):
71
+ decoded = data.decode("utf-8", errors="replace")
72
+ return decoded.removeprefix("").encode()
73
+
74
+
44
75
  def _read_tabular(path: Path, separator: str = ",") -> pl.DataFrame:
45
76
  """CSV/TSV read with a retry fallback.
46
77
 
47
78
  Inference looks at up to 10,000 rows; dtype-incompatible values can
48
79
  still appear later in a large file and fail parsing. When that happens,
49
80
  re-read with zero-length inference so every column stays String instead
50
- of erroring out."""
81
+ of erroring out. The same retry also covers a non-UTF-8 encoding: the
82
+ sniffer (`_decoded_text`) transcoded the bytes to UTF-8."""
51
83
  try:
52
84
  return pl.read_csv(path, separator=separator, infer_schema_length=10_000)
53
85
  except pl.exceptions.PolarsError:
54
- return pl.read_csv(path, separator=separator, infer_schema_length=0)
86
+ data = _decoded_text(path.read_bytes())
87
+ try:
88
+ return pl.read_csv(
89
+ BytesIO(data), separator=separator, infer_schema_length=10_000
90
+ )
91
+ except pl.exceptions.PolarsError:
92
+ return pl.read_csv(
93
+ BytesIO(data), separator=separator, infer_schema_length=0
94
+ )
55
95
 
56
96
 
57
97
  def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
@@ -61,7 +101,10 @@ def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
61
101
  if fmt == "tsv":
62
102
  return _read_tabular(path, separator="\t")
63
103
  if fmt == "jsonl":
64
- return pl.read_ndjson(path)
104
+ try:
105
+ return pl.read_ndjson(path)
106
+ except pl.exceptions.PolarsError:
107
+ return pl.read_ndjson(BytesIO(_decoded_text(path.read_bytes())))
65
108
  if fmt == "parquet":
66
109
  return pl.read_parquet(path)
67
110
  if fmt == "xlsx":