normalize-tabular-data 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,187 @@
1
+ """Format dispatch: suffix -> polars read/write. No Textual imports."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ from pathlib import Path
8
+
9
+ import polars as pl
10
+
11
+ READ_SUFFIXES: dict[str, str] = {
12
+ ".csv": "csv",
13
+ ".tsv": "tsv",
14
+ ".txt": "tsv",
15
+ ".jsonl": "jsonl",
16
+ ".ndjson": "jsonl",
17
+ ".parquet": "parquet",
18
+ ".pq": "parquet",
19
+ ".parq": "parquet",
20
+ ".xlsx": "xlsx",
21
+ ".xls": "xlsx",
22
+ }
23
+
24
+ WRITE_SUFFIXES: dict[str, str] = {k: v for k, v in READ_SUFFIXES.items() if k != ".xls"}
25
+
26
+ FORMAT_SUFFIX: dict[str, str] = {
27
+ "csv": ".csv",
28
+ "tsv": ".tsv",
29
+ "jsonl": ".jsonl",
30
+ "parquet": ".parquet",
31
+ "xlsx": ".xlsx",
32
+ }
33
+
34
+
35
+ def detect_format(path: Path) -> str:
36
+ suffix = path.suffix.lower()
37
+ fmt = READ_SUFFIXES.get(suffix)
38
+ if fmt is None:
39
+ known = ", ".join(sorted(READ_SUFFIXES))
40
+ raise ValueError(f"Unsupported file type {suffix!r} (known: {known})")
41
+ return fmt
42
+
43
+
44
+ def _read_tabular(path: Path, separator: str = ",") -> pl.DataFrame:
45
+ """CSV/TSV read with a retry fallback.
46
+
47
+ Inference looks at up to 10,000 rows; dtype-incompatible values can
48
+ still appear later in a large file and fail parsing. When that happens,
49
+ re-read with zero-length inference so every column stays String instead
50
+ of erroring out."""
51
+ try:
52
+ return pl.read_csv(path, separator=separator, infer_schema_length=10_000)
53
+ except pl.exceptions.PolarsError:
54
+ return pl.read_csv(path, separator=separator, infer_schema_length=0)
55
+
56
+
57
+ def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
58
+ """Read one table. For xlsx, `sheet` picks a sheet; default is the first."""
59
+ if fmt == "csv":
60
+ return _read_tabular(path)
61
+ if fmt == "tsv":
62
+ return _read_tabular(path, separator="\t")
63
+ if fmt == "jsonl":
64
+ return pl.read_ndjson(path)
65
+ if fmt == "parquet":
66
+ return pl.read_parquet(path)
67
+ if fmt == "xlsx":
68
+ if sheet is None:
69
+ sheets = excel_sheets(path)
70
+ sheet = sheets[0] if sheets else "Sheet1"
71
+ return pl.read_excel(path, sheet_name=sheet)
72
+ raise ValueError(f"Unknown format {fmt!r}")
73
+
74
+
75
+ def excel_sheets(path: Path) -> list[str]:
76
+ """Sheet names of a workbook, in order (fastexcel/calamine)."""
77
+ import fastexcel
78
+
79
+ return list(fastexcel.read_excel(path).sheet_names)
80
+
81
+
82
+ def write_table(
83
+ df: pl.DataFrame,
84
+ path: Path,
85
+ fmt: str,
86
+ sheet_name: str = "data",
87
+ ) -> None:
88
+ if fmt == "csv":
89
+ df.write_csv(path)
90
+ elif fmt == "tsv":
91
+ df.write_csv(path, separator="\t")
92
+ elif fmt == "jsonl":
93
+ df.write_ndjson(path)
94
+ elif fmt == "parquet":
95
+ df.write_parquet(path)
96
+ elif fmt == "xlsx":
97
+ df.write_excel(path, worksheet=sheet_name, autofit=True, freeze_panes="B2")
98
+ else:
99
+ raise ValueError(f"Unknown format {fmt!r}")
100
+
101
+
102
+ # --- operation scripts --------------------------------------------------------
103
+
104
+ SCRIPT_SUFFIX = ".ntd"
105
+
106
+
107
+ def script_text(steps: list[tuple[str, dict]], header_fields: dict[str, str]) -> str:
108
+ """Human-readable ASCII text, one operation description per line.
109
+
110
+ Every operation is `<key>(<param>=<json value>, ...)` so the lines stay
111
+ readable while remaining mechanically parseable for a future
112
+ "apply a script" capability. `header_fields` become leading `#`
113
+ comment lines (source file, save time, ...)"""
114
+ lines = ["# normalize-tabular-data script"]
115
+ lines += [f"# {text}" for text in header_fields.values() if text]
116
+ lines += [
117
+ f"{key}("
118
+ + ", ".join(f"{param}={json.dumps(value)}" for param, value in params.items())
119
+ + ")"
120
+ for key, params in steps
121
+ ]
122
+ return "\n".join(lines) + "\n"
123
+
124
+
125
+ def write_script(
126
+ path: Path,
127
+ steps: list[tuple[str, dict]],
128
+ header_fields: dict[str, str],
129
+ ) -> None:
130
+ """Write the script as plain ASCII (non-ASCII input becomes '?')."""
131
+ path.write_text(
132
+ script_text(steps, header_fields), encoding="ascii", errors="replace"
133
+ )
134
+
135
+
136
+ _SCRIPT_LINE_RE = re.compile(r"([A-Za-z_][A-Za-z0-9_]*)\((.*)\)")
137
+ _PARAM_NAME_RE = re.compile(r"([A-Za-z_][A-Za-z0-9_]*)\s*=\s*")
138
+
139
+
140
+ def parse_script_line(line: str) -> tuple[str, dict]:
141
+ """Parse one `key(param=<json value>, ...)` operation description.
142
+
143
+ Values are plain JSON literals, so each parameter is read with a
144
+ JSON decoder (no evaluation). Raises ValueError on anything else."""
145
+ match = _SCRIPT_LINE_RE.fullmatch(line.strip())
146
+ if match is None:
147
+ raise ValueError(f"not an operation line: {line!r}")
148
+ key, args = match.group(1), match.group(2)
149
+ decoder = json.JSONDecoder()
150
+ params: dict = {}
151
+ pos = 0
152
+ n = len(args)
153
+ while pos < n:
154
+ name_match = _PARAM_NAME_RE.match(args, pos)
155
+ if name_match is None:
156
+ raise ValueError(f"missing parameter value in: {line!r}")
157
+ name = name_match.group(1)
158
+ pos = name_match.end()
159
+ try:
160
+ params[name], pos = decoder.raw_decode(args, pos)
161
+ except json.JSONDecodeError:
162
+ raise ValueError(f"bad value for {name!r} in: {line!r}") from None
163
+ while pos < n and args[pos] in " \t":
164
+ pos += 1
165
+ if pos < n and args[pos] == ",":
166
+ pos += 1
167
+ while pos < n and args[pos] in " \t":
168
+ pos += 1
169
+ elif pos < n:
170
+ raise ValueError(f"expected ',' between parameters in: {line!r}")
171
+ return key, params
172
+
173
+
174
+ def read_script(path: Path) -> list[tuple[str, dict]]:
175
+ """Read an operation script: one (key, params) step per non-comment line."""
176
+ steps: list[tuple[str, dict]] = []
177
+ for lineno, raw in enumerate(
178
+ path.read_text(encoding="ascii", errors="replace").splitlines(), 1
179
+ ):
180
+ line = raw.strip()
181
+ if not line or line.startswith("#"):
182
+ continue
183
+ try:
184
+ steps.append(parse_script_line(line))
185
+ except ValueError as exc:
186
+ raise ValueError(f"{path.name} line {lineno}: {exc}") from exc
187
+ return steps
@@ -0,0 +1,333 @@
1
+ """Normalization engine: operation registry and re-computing pipeline.
2
+
3
+ Pure polars — no Textual imports here, so the whole engine is testable
4
+ without a terminal.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from collections.abc import Callable
10
+ from dataclasses import dataclass, field
11
+ from typing import Any, Literal
12
+
13
+ import date_parser
14
+ import polars as pl
15
+
16
+ ParamKind = Literal["column_multi", "column", "text", "number", "choice", "bool"]
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class ParamSpec:
21
+ """One parameter of an operation, consumed by the UI to build a dialog."""
22
+
23
+ name: str
24
+ kind: ParamKind
25
+ title: str
26
+ default: Any = None
27
+ choices: tuple[str, ...] | None = None
28
+ min_value: int | None = None
29
+ help: str = ""
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class Operation:
34
+ key: str
35
+ title: str
36
+ params: tuple[ParamSpec, ...]
37
+ apply: Callable[[pl.DataFrame, dict[str, Any]], pl.DataFrame]
38
+ # designated chooser hotkey letter; "" lets the chooser pick the first
39
+ # unused letter in the title
40
+ hotkey: str = ""
41
+
42
+
43
+ # --- apply functions -------------------------------------------------------
44
+
45
+
46
+ def _apply_date_normalize(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
47
+ col: str = p["column"]
48
+ series = df.get_column(col)
49
+ if series.dtype != pl.String:
50
+ series = series.cast(pl.String)
51
+ # Datetime("ns"), unparseable -> null
52
+ return df.with_columns(date_parser.parse_series(series).alias(col))
53
+
54
+
55
+ def _apply_trim_collapse(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
56
+ cols: list[str] = p["columns"]
57
+ return df.with_columns(
58
+ pl.col(c).cast(pl.String).str.strip_chars().str.replace_all(r"\s+", " ")
59
+ for c in cols
60
+ )
61
+
62
+
63
+ def _apply_dedup_rows(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
64
+ cols: list[str] = p["columns"]
65
+ keep: str = p["keep"]
66
+ return df.unique(
67
+ subset=cols or None,
68
+ keep="first" if keep == "first" else "last",
69
+ )
70
+
71
+
72
+ def _apply_drop_columns(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
73
+ return df.drop(p["columns"])
74
+
75
+
76
+ def _apply_combine_columns(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
77
+ cols: list[str] = p["columns"]
78
+ sep: str = p["separator"]
79
+ name: str = p["new_name"]
80
+ return df.with_columns(
81
+ pl.concat_str(
82
+ [pl.col(c).cast(pl.String) for c in cols],
83
+ separator=sep,
84
+ ignore_nulls=True,
85
+ ).alias(name)
86
+ )
87
+
88
+
89
+ def _apply_split_column(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
90
+ col: str = p["column"]
91
+ delim: str = p["delimiter"]
92
+ rest = df.drop(col)
93
+ # Always strip edge whitespace first; a space delimiter additionally
94
+ # collapses internal whitespace runs so one-or-many spaces/tabs split
95
+ # the same way.
96
+ cleaned = pl.col(col).cast(pl.String).str.strip_chars()
97
+ if delim.strip() == "" or set(delim) <= set(" \t\r\n\f\v"):
98
+ cleaned = cleaned.str.replace_all(r"\s+", " ").replace("", None)
99
+ delim = " "
100
+ split = df.select(cleaned.str.split(delim).alias("__split"))
101
+ # part count comes from the data: the longest actual split for this
102
+ # delimiter; short rows pad with null
103
+ width = split.select(pl.col("__split").list.len().max()).item() or 1
104
+ names = [f"{col}_{i + 1}" for i in range(width)]
105
+ parts = split.select(
106
+ pl.col("__split").list.to_struct(
107
+ fields=names,
108
+ upper_bound=width,
109
+ )
110
+ )
111
+ pieces = parts.get_column("__split").struct.unnest()
112
+ return rest.hstack(pieces) if rest.width else pieces
113
+
114
+
115
+ # --- registry ---------------------------------------------------------------
116
+
117
+ OPS: tuple[Operation, ...] = (
118
+ Operation(
119
+ key="date_normalize",
120
+ title="Normalize dates",
121
+ hotkey="d",
122
+ params=(
123
+ ParamSpec(
124
+ "column",
125
+ "column",
126
+ "Date column",
127
+ help="Any input format; unparseable -> null",
128
+ ),
129
+ ),
130
+ apply=_apply_date_normalize,
131
+ ),
132
+ Operation(
133
+ key="trim_collapse",
134
+ title="Trim whitespace",
135
+ hotkey="w",
136
+ params=(
137
+ ParamSpec(
138
+ "columns",
139
+ "column_multi",
140
+ "Columns",
141
+ help="Strip edges, collapse internal runs to one space",
142
+ ),
143
+ ),
144
+ apply=_apply_trim_collapse,
145
+ ),
146
+ Operation(
147
+ key="dedup_rows",
148
+ title="Deduplicate rows",
149
+ hotkey="p",
150
+ params=(
151
+ ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
152
+ ParamSpec(
153
+ "keep",
154
+ "choice",
155
+ "Keep",
156
+ default="first",
157
+ choices=("first", "last"),
158
+ ),
159
+ ),
160
+ apply=_apply_dedup_rows,
161
+ ),
162
+ Operation(
163
+ key="combine_columns",
164
+ title="Combine columns",
165
+ hotkey="c",
166
+ params=(
167
+ ParamSpec("columns", "column_multi", "Columns (>= 2)"),
168
+ ParamSpec("separator", "text", "Separator", default=" "),
169
+ ParamSpec("new_name", "text", "New column name"),
170
+ ),
171
+ apply=_apply_combine_columns,
172
+ ),
173
+ Operation(
174
+ key="split_column",
175
+ title="Split column",
176
+ hotkey="s",
177
+ params=(
178
+ ParamSpec("column", "column", "Column to split"),
179
+ ParamSpec(
180
+ "delimiter",
181
+ "text",
182
+ "Delimiter",
183
+ default="",
184
+ help=(
185
+ "Default split is on whitespace. "
186
+ "If specified, split on exact characters given."
187
+ ),
188
+ ),
189
+ ),
190
+ apply=_apply_split_column,
191
+ ),
192
+ Operation(
193
+ key="remove_columns",
194
+ title="Remove columns",
195
+ hotkey="r",
196
+ params=(
197
+ ParamSpec(
198
+ "columns",
199
+ "column_multi",
200
+ "Columns to remove",
201
+ help="Remove the selected columns entirely",
202
+ ),
203
+ ),
204
+ apply=_apply_drop_columns,
205
+ ),
206
+ )
207
+
208
+ OP_REGISTRY: dict[str, Operation] = {op.key: op for op in OPS}
209
+
210
+ # internal op (not in the chooser): renames one column via sample header clicks
211
+ RENAME_OP = Operation(
212
+ key="rename_single",
213
+ title="Rename column",
214
+ params=(
215
+ ParamSpec("column", "column", "Column"),
216
+ ParamSpec("new_name", "text", "New name"),
217
+ ),
218
+ apply=lambda df, p: df.rename({p["column"]: p["new_name"]}),
219
+ )
220
+
221
+ # The script player resolves against this: every chooser op plus the internal
222
+ # rename op (header-click renames are logged under the key "rename_single",
223
+ # so saved scripts contain those lines and have to play back).
224
+ PLAY_REGISTRY: dict[str, Operation] = {**OP_REGISTRY, RENAME_OP.key: RENAME_OP}
225
+
226
+
227
+ # --- pipeline ---------------------------------------------------------------
228
+
229
+
230
+ @dataclass
231
+ class AppliedOp:
232
+ op: Operation
233
+ params: dict[str, Any]
234
+
235
+
236
+ @dataclass
237
+ class Pipeline:
238
+ """Original DataFrame + ordered applied ops; every change re-folds from source."""
239
+
240
+ source: pl.DataFrame
241
+ applied: list[AppliedOp] = field(default_factory=list)
242
+ redo_stack: list[AppliedOp] = field(default_factory=list)
243
+
244
+ def current(self) -> pl.DataFrame:
245
+ df = self.source
246
+ for step in self.applied:
247
+ df = step.op.apply(df, step.params)
248
+ return df
249
+
250
+ def validate(self, op: Operation, params: dict[str, Any]) -> str | None:
251
+ cols = list(self.current().columns)
252
+ for spec in op.params:
253
+ val = params.get(spec.name)
254
+ if spec.kind == "column":
255
+ if not isinstance(val, str) or val not in cols:
256
+ return f"{op.title}: pick a column"
257
+ elif spec.kind == "column_multi":
258
+ picked = [c for c in (val or []) if c in cols]
259
+ if not picked:
260
+ return f"{op.title}: pick at least one column"
261
+ need = 2 if op.key == "combine_columns" else 1
262
+ if len(set(picked)) < need:
263
+ return f"{op.title}: pick at least {need} columns"
264
+ elif spec.kind == "text":
265
+ # split_column's blank delimiter is allowed: it means
266
+ # "split on runs of whitespace"
267
+ if op.key == "combine_columns" and spec.name == "new_name" and not val:
268
+ return f"{op.title}: new name required"
269
+ if (
270
+ op.key == "combine_columns"
271
+ and spec.name == "new_name"
272
+ and val in cols
273
+ ):
274
+ return f"{op.title}: {val!r} already exists"
275
+ elif spec.kind == "number":
276
+ if val is None:
277
+ return f"{op.title}: set a value for {spec.title}"
278
+ if spec.min_value is not None and val < spec.min_value:
279
+ return f"{op.title}: {spec.title} must be >= {spec.min_value}"
280
+ return None
281
+
282
+ def apply(self, op: Operation, params: dict[str, Any]) -> None:
283
+ if (err := self.validate(op, params)) is not None:
284
+ raise ValueError(err)
285
+ self.applied.append(AppliedOp(op, dict(params)))
286
+ self.redo_stack.clear()
287
+
288
+ def undo(self) -> bool:
289
+ if not self.applied:
290
+ return False
291
+ self.redo_stack.append(self.applied.pop())
292
+ return True
293
+
294
+ def redo(self) -> bool:
295
+ if not self.redo_stack:
296
+ return False
297
+ self.applied.append(self.redo_stack.pop())
298
+ return True
299
+
300
+ def step_summary(self) -> list[str]:
301
+ out = []
302
+ for step in self.applied:
303
+ bits = " ".join(
304
+ f"{k}={v}" for k, v in step.params.items() if v not in (None, ())
305
+ )
306
+ out.append(f"{step.op.key}({bits})" if bits else step.op.key)
307
+ return out
308
+
309
+
310
+ # --- column analysis ---------------------------------------------------------
311
+
312
+
313
+ @dataclass
314
+ class ColumnInfo:
315
+ name: str
316
+ dtype: str
317
+ null_count: int
318
+ date_candidate: bool = False
319
+
320
+
321
+ def analyze_column(df: pl.DataFrame, col: str) -> ColumnInfo:
322
+ """dtype, null count, and a date-candidate flag from a 200-row sample."""
323
+ s = df.get_column(col)
324
+ info = ColumnInfo(name=col, dtype=str(s.dtype), null_count=int(s.null_count()))
325
+ if s.dtype == pl.String:
326
+ sample = (
327
+ s.drop_nulls().str.strip_chars().replace("", None).drop_nulls().head(200)
328
+ )
329
+ if sample.len() >= 3:
330
+ parsed = date_parser.parse_series(sample)
331
+ ratio = 1.0 - (parsed.null_count() / sample.len())
332
+ info.date_candidate = ratio >= 0.5
333
+ return info