normalize-tabular-data 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- normalize_tabular_data/__init__.py +18 -0
- normalize_tabular_data/__main__.py +4 -0
- normalize_tabular_data/app.py +644 -0
- normalize_tabular_data/io.py +187 -0
- normalize_tabular_data/ops.py +333 -0
- normalize_tabular_data/screens.py +568 -0
- normalize_tabular_data/widgets.py +227 -0
- normalize_tabular_data-0.1.0.dist-info/METADATA +164 -0
- normalize_tabular_data-0.1.0.dist-info/RECORD +12 -0
- normalize_tabular_data-0.1.0.dist-info/WHEEL +4 -0
- normalize_tabular_data-0.1.0.dist-info/entry_points.txt +3 -0
- normalize_tabular_data-0.1.0.dist-info/licenses/LICENSE +24 -0
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""Format dispatch: suffix -> polars read/write. No Textual imports."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
|
|
11
|
+
READ_SUFFIXES: dict[str, str] = {
|
|
12
|
+
".csv": "csv",
|
|
13
|
+
".tsv": "tsv",
|
|
14
|
+
".txt": "tsv",
|
|
15
|
+
".jsonl": "jsonl",
|
|
16
|
+
".ndjson": "jsonl",
|
|
17
|
+
".parquet": "parquet",
|
|
18
|
+
".pq": "parquet",
|
|
19
|
+
".parq": "parquet",
|
|
20
|
+
".xlsx": "xlsx",
|
|
21
|
+
".xls": "xlsx",
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
WRITE_SUFFIXES: dict[str, str] = {k: v for k, v in READ_SUFFIXES.items() if k != ".xls"}
|
|
25
|
+
|
|
26
|
+
FORMAT_SUFFIX: dict[str, str] = {
|
|
27
|
+
"csv": ".csv",
|
|
28
|
+
"tsv": ".tsv",
|
|
29
|
+
"jsonl": ".jsonl",
|
|
30
|
+
"parquet": ".parquet",
|
|
31
|
+
"xlsx": ".xlsx",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def detect_format(path: Path) -> str:
|
|
36
|
+
suffix = path.suffix.lower()
|
|
37
|
+
fmt = READ_SUFFIXES.get(suffix)
|
|
38
|
+
if fmt is None:
|
|
39
|
+
known = ", ".join(sorted(READ_SUFFIXES))
|
|
40
|
+
raise ValueError(f"Unsupported file type {suffix!r} (known: {known})")
|
|
41
|
+
return fmt
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _read_tabular(path: Path, separator: str = ",") -> pl.DataFrame:
|
|
45
|
+
"""CSV/TSV read with a retry fallback.
|
|
46
|
+
|
|
47
|
+
Inference looks at up to 10,000 rows; dtype-incompatible values can
|
|
48
|
+
still appear later in a large file and fail parsing. When that happens,
|
|
49
|
+
re-read with zero-length inference so every column stays String instead
|
|
50
|
+
of erroring out."""
|
|
51
|
+
try:
|
|
52
|
+
return pl.read_csv(path, separator=separator, infer_schema_length=10_000)
|
|
53
|
+
except pl.exceptions.PolarsError:
|
|
54
|
+
return pl.read_csv(path, separator=separator, infer_schema_length=0)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def read_table(path: Path, fmt: str, sheet: str | None = None) -> pl.DataFrame:
|
|
58
|
+
"""Read one table. For xlsx, `sheet` picks a sheet; default is the first."""
|
|
59
|
+
if fmt == "csv":
|
|
60
|
+
return _read_tabular(path)
|
|
61
|
+
if fmt == "tsv":
|
|
62
|
+
return _read_tabular(path, separator="\t")
|
|
63
|
+
if fmt == "jsonl":
|
|
64
|
+
return pl.read_ndjson(path)
|
|
65
|
+
if fmt == "parquet":
|
|
66
|
+
return pl.read_parquet(path)
|
|
67
|
+
if fmt == "xlsx":
|
|
68
|
+
if sheet is None:
|
|
69
|
+
sheets = excel_sheets(path)
|
|
70
|
+
sheet = sheets[0] if sheets else "Sheet1"
|
|
71
|
+
return pl.read_excel(path, sheet_name=sheet)
|
|
72
|
+
raise ValueError(f"Unknown format {fmt!r}")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def excel_sheets(path: Path) -> list[str]:
|
|
76
|
+
"""Sheet names of a workbook, in order (fastexcel/calamine)."""
|
|
77
|
+
import fastexcel
|
|
78
|
+
|
|
79
|
+
return list(fastexcel.read_excel(path).sheet_names)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def write_table(
|
|
83
|
+
df: pl.DataFrame,
|
|
84
|
+
path: Path,
|
|
85
|
+
fmt: str,
|
|
86
|
+
sheet_name: str = "data",
|
|
87
|
+
) -> None:
|
|
88
|
+
if fmt == "csv":
|
|
89
|
+
df.write_csv(path)
|
|
90
|
+
elif fmt == "tsv":
|
|
91
|
+
df.write_csv(path, separator="\t")
|
|
92
|
+
elif fmt == "jsonl":
|
|
93
|
+
df.write_ndjson(path)
|
|
94
|
+
elif fmt == "parquet":
|
|
95
|
+
df.write_parquet(path)
|
|
96
|
+
elif fmt == "xlsx":
|
|
97
|
+
df.write_excel(path, worksheet=sheet_name, autofit=True, freeze_panes="B2")
|
|
98
|
+
else:
|
|
99
|
+
raise ValueError(f"Unknown format {fmt!r}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# --- operation scripts --------------------------------------------------------
|
|
103
|
+
|
|
104
|
+
SCRIPT_SUFFIX = ".ntd"
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def script_text(steps: list[tuple[str, dict]], header_fields: dict[str, str]) -> str:
|
|
108
|
+
"""Human-readable ASCII text, one operation description per line.
|
|
109
|
+
|
|
110
|
+
Every operation is `<key>(<param>=<json value>, ...)` so the lines stay
|
|
111
|
+
readable while remaining mechanically parseable for a future
|
|
112
|
+
"apply a script" capability. `header_fields` become leading `#`
|
|
113
|
+
comment lines (source file, save time, ...)"""
|
|
114
|
+
lines = ["# normalize-tabular-data script"]
|
|
115
|
+
lines += [f"# {text}" for text in header_fields.values() if text]
|
|
116
|
+
lines += [
|
|
117
|
+
f"{key}("
|
|
118
|
+
+ ", ".join(f"{param}={json.dumps(value)}" for param, value in params.items())
|
|
119
|
+
+ ")"
|
|
120
|
+
for key, params in steps
|
|
121
|
+
]
|
|
122
|
+
return "\n".join(lines) + "\n"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def write_script(
|
|
126
|
+
path: Path,
|
|
127
|
+
steps: list[tuple[str, dict]],
|
|
128
|
+
header_fields: dict[str, str],
|
|
129
|
+
) -> None:
|
|
130
|
+
"""Write the script as plain ASCII (non-ASCII input becomes '?')."""
|
|
131
|
+
path.write_text(
|
|
132
|
+
script_text(steps, header_fields), encoding="ascii", errors="replace"
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
_SCRIPT_LINE_RE = re.compile(r"([A-Za-z_][A-Za-z0-9_]*)\((.*)\)")
|
|
137
|
+
_PARAM_NAME_RE = re.compile(r"([A-Za-z_][A-Za-z0-9_]*)\s*=\s*")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def parse_script_line(line: str) -> tuple[str, dict]:
|
|
141
|
+
"""Parse one `key(param=<json value>, ...)` operation description.
|
|
142
|
+
|
|
143
|
+
Values are plain JSON literals, so each parameter is read with a
|
|
144
|
+
JSON decoder (no evaluation). Raises ValueError on anything else."""
|
|
145
|
+
match = _SCRIPT_LINE_RE.fullmatch(line.strip())
|
|
146
|
+
if match is None:
|
|
147
|
+
raise ValueError(f"not an operation line: {line!r}")
|
|
148
|
+
key, args = match.group(1), match.group(2)
|
|
149
|
+
decoder = json.JSONDecoder()
|
|
150
|
+
params: dict = {}
|
|
151
|
+
pos = 0
|
|
152
|
+
n = len(args)
|
|
153
|
+
while pos < n:
|
|
154
|
+
name_match = _PARAM_NAME_RE.match(args, pos)
|
|
155
|
+
if name_match is None:
|
|
156
|
+
raise ValueError(f"missing parameter value in: {line!r}")
|
|
157
|
+
name = name_match.group(1)
|
|
158
|
+
pos = name_match.end()
|
|
159
|
+
try:
|
|
160
|
+
params[name], pos = decoder.raw_decode(args, pos)
|
|
161
|
+
except json.JSONDecodeError:
|
|
162
|
+
raise ValueError(f"bad value for {name!r} in: {line!r}") from None
|
|
163
|
+
while pos < n and args[pos] in " \t":
|
|
164
|
+
pos += 1
|
|
165
|
+
if pos < n and args[pos] == ",":
|
|
166
|
+
pos += 1
|
|
167
|
+
while pos < n and args[pos] in " \t":
|
|
168
|
+
pos += 1
|
|
169
|
+
elif pos < n:
|
|
170
|
+
raise ValueError(f"expected ',' between parameters in: {line!r}")
|
|
171
|
+
return key, params
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def read_script(path: Path) -> list[tuple[str, dict]]:
|
|
175
|
+
"""Read an operation script: one (key, params) step per non-comment line."""
|
|
176
|
+
steps: list[tuple[str, dict]] = []
|
|
177
|
+
for lineno, raw in enumerate(
|
|
178
|
+
path.read_text(encoding="ascii", errors="replace").splitlines(), 1
|
|
179
|
+
):
|
|
180
|
+
line = raw.strip()
|
|
181
|
+
if not line or line.startswith("#"):
|
|
182
|
+
continue
|
|
183
|
+
try:
|
|
184
|
+
steps.append(parse_script_line(line))
|
|
185
|
+
except ValueError as exc:
|
|
186
|
+
raise ValueError(f"{path.name} line {lineno}: {exc}") from exc
|
|
187
|
+
return steps
|
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
"""Normalization engine: operation registry and re-computing pipeline.
|
|
2
|
+
|
|
3
|
+
Pure polars — no Textual imports here, so the whole engine is testable
|
|
4
|
+
without a terminal.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Literal
|
|
12
|
+
|
|
13
|
+
import date_parser
|
|
14
|
+
import polars as pl
|
|
15
|
+
|
|
16
|
+
ParamKind = Literal["column_multi", "column", "text", "number", "choice", "bool"]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class ParamSpec:
|
|
21
|
+
"""One parameter of an operation, consumed by the UI to build a dialog."""
|
|
22
|
+
|
|
23
|
+
name: str
|
|
24
|
+
kind: ParamKind
|
|
25
|
+
title: str
|
|
26
|
+
default: Any = None
|
|
27
|
+
choices: tuple[str, ...] | None = None
|
|
28
|
+
min_value: int | None = None
|
|
29
|
+
help: str = ""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class Operation:
|
|
34
|
+
key: str
|
|
35
|
+
title: str
|
|
36
|
+
params: tuple[ParamSpec, ...]
|
|
37
|
+
apply: Callable[[pl.DataFrame, dict[str, Any]], pl.DataFrame]
|
|
38
|
+
# designated chooser hotkey letter; "" lets the chooser pick the first
|
|
39
|
+
# unused letter in the title
|
|
40
|
+
hotkey: str = ""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# --- apply functions -------------------------------------------------------
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _apply_date_normalize(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
47
|
+
col: str = p["column"]
|
|
48
|
+
series = df.get_column(col)
|
|
49
|
+
if series.dtype != pl.String:
|
|
50
|
+
series = series.cast(pl.String)
|
|
51
|
+
# Datetime("ns"), unparseable -> null
|
|
52
|
+
return df.with_columns(date_parser.parse_series(series).alias(col))
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _apply_trim_collapse(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
56
|
+
cols: list[str] = p["columns"]
|
|
57
|
+
return df.with_columns(
|
|
58
|
+
pl.col(c).cast(pl.String).str.strip_chars().str.replace_all(r"\s+", " ")
|
|
59
|
+
for c in cols
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _apply_dedup_rows(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
64
|
+
cols: list[str] = p["columns"]
|
|
65
|
+
keep: str = p["keep"]
|
|
66
|
+
return df.unique(
|
|
67
|
+
subset=cols or None,
|
|
68
|
+
keep="first" if keep == "first" else "last",
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _apply_drop_columns(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
73
|
+
return df.drop(p["columns"])
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _apply_combine_columns(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
77
|
+
cols: list[str] = p["columns"]
|
|
78
|
+
sep: str = p["separator"]
|
|
79
|
+
name: str = p["new_name"]
|
|
80
|
+
return df.with_columns(
|
|
81
|
+
pl.concat_str(
|
|
82
|
+
[pl.col(c).cast(pl.String) for c in cols],
|
|
83
|
+
separator=sep,
|
|
84
|
+
ignore_nulls=True,
|
|
85
|
+
).alias(name)
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _apply_split_column(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
90
|
+
col: str = p["column"]
|
|
91
|
+
delim: str = p["delimiter"]
|
|
92
|
+
rest = df.drop(col)
|
|
93
|
+
# Always strip edge whitespace first; a space delimiter additionally
|
|
94
|
+
# collapses internal whitespace runs so one-or-many spaces/tabs split
|
|
95
|
+
# the same way.
|
|
96
|
+
cleaned = pl.col(col).cast(pl.String).str.strip_chars()
|
|
97
|
+
if delim.strip() == "" or set(delim) <= set(" \t\r\n\f\v"):
|
|
98
|
+
cleaned = cleaned.str.replace_all(r"\s+", " ").replace("", None)
|
|
99
|
+
delim = " "
|
|
100
|
+
split = df.select(cleaned.str.split(delim).alias("__split"))
|
|
101
|
+
# part count comes from the data: the longest actual split for this
|
|
102
|
+
# delimiter; short rows pad with null
|
|
103
|
+
width = split.select(pl.col("__split").list.len().max()).item() or 1
|
|
104
|
+
names = [f"{col}_{i + 1}" for i in range(width)]
|
|
105
|
+
parts = split.select(
|
|
106
|
+
pl.col("__split").list.to_struct(
|
|
107
|
+
fields=names,
|
|
108
|
+
upper_bound=width,
|
|
109
|
+
)
|
|
110
|
+
)
|
|
111
|
+
pieces = parts.get_column("__split").struct.unnest()
|
|
112
|
+
return rest.hstack(pieces) if rest.width else pieces
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
# --- registry ---------------------------------------------------------------
|
|
116
|
+
|
|
117
|
+
OPS: tuple[Operation, ...] = (
|
|
118
|
+
Operation(
|
|
119
|
+
key="date_normalize",
|
|
120
|
+
title="Normalize dates",
|
|
121
|
+
hotkey="d",
|
|
122
|
+
params=(
|
|
123
|
+
ParamSpec(
|
|
124
|
+
"column",
|
|
125
|
+
"column",
|
|
126
|
+
"Date column",
|
|
127
|
+
help="Any input format; unparseable -> null",
|
|
128
|
+
),
|
|
129
|
+
),
|
|
130
|
+
apply=_apply_date_normalize,
|
|
131
|
+
),
|
|
132
|
+
Operation(
|
|
133
|
+
key="trim_collapse",
|
|
134
|
+
title="Trim whitespace",
|
|
135
|
+
hotkey="w",
|
|
136
|
+
params=(
|
|
137
|
+
ParamSpec(
|
|
138
|
+
"columns",
|
|
139
|
+
"column_multi",
|
|
140
|
+
"Columns",
|
|
141
|
+
help="Strip edges, collapse internal runs to one space",
|
|
142
|
+
),
|
|
143
|
+
),
|
|
144
|
+
apply=_apply_trim_collapse,
|
|
145
|
+
),
|
|
146
|
+
Operation(
|
|
147
|
+
key="dedup_rows",
|
|
148
|
+
title="Deduplicate rows",
|
|
149
|
+
hotkey="p",
|
|
150
|
+
params=(
|
|
151
|
+
ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
|
|
152
|
+
ParamSpec(
|
|
153
|
+
"keep",
|
|
154
|
+
"choice",
|
|
155
|
+
"Keep",
|
|
156
|
+
default="first",
|
|
157
|
+
choices=("first", "last"),
|
|
158
|
+
),
|
|
159
|
+
),
|
|
160
|
+
apply=_apply_dedup_rows,
|
|
161
|
+
),
|
|
162
|
+
Operation(
|
|
163
|
+
key="combine_columns",
|
|
164
|
+
title="Combine columns",
|
|
165
|
+
hotkey="c",
|
|
166
|
+
params=(
|
|
167
|
+
ParamSpec("columns", "column_multi", "Columns (>= 2)"),
|
|
168
|
+
ParamSpec("separator", "text", "Separator", default=" "),
|
|
169
|
+
ParamSpec("new_name", "text", "New column name"),
|
|
170
|
+
),
|
|
171
|
+
apply=_apply_combine_columns,
|
|
172
|
+
),
|
|
173
|
+
Operation(
|
|
174
|
+
key="split_column",
|
|
175
|
+
title="Split column",
|
|
176
|
+
hotkey="s",
|
|
177
|
+
params=(
|
|
178
|
+
ParamSpec("column", "column", "Column to split"),
|
|
179
|
+
ParamSpec(
|
|
180
|
+
"delimiter",
|
|
181
|
+
"text",
|
|
182
|
+
"Delimiter",
|
|
183
|
+
default="",
|
|
184
|
+
help=(
|
|
185
|
+
"Default split is on whitespace. "
|
|
186
|
+
"If specified, split on exact characters given."
|
|
187
|
+
),
|
|
188
|
+
),
|
|
189
|
+
),
|
|
190
|
+
apply=_apply_split_column,
|
|
191
|
+
),
|
|
192
|
+
Operation(
|
|
193
|
+
key="remove_columns",
|
|
194
|
+
title="Remove columns",
|
|
195
|
+
hotkey="r",
|
|
196
|
+
params=(
|
|
197
|
+
ParamSpec(
|
|
198
|
+
"columns",
|
|
199
|
+
"column_multi",
|
|
200
|
+
"Columns to remove",
|
|
201
|
+
help="Remove the selected columns entirely",
|
|
202
|
+
),
|
|
203
|
+
),
|
|
204
|
+
apply=_apply_drop_columns,
|
|
205
|
+
),
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
OP_REGISTRY: dict[str, Operation] = {op.key: op for op in OPS}
|
|
209
|
+
|
|
210
|
+
# internal op (not in the chooser): renames one column via sample header clicks
|
|
211
|
+
RENAME_OP = Operation(
|
|
212
|
+
key="rename_single",
|
|
213
|
+
title="Rename column",
|
|
214
|
+
params=(
|
|
215
|
+
ParamSpec("column", "column", "Column"),
|
|
216
|
+
ParamSpec("new_name", "text", "New name"),
|
|
217
|
+
),
|
|
218
|
+
apply=lambda df, p: df.rename({p["column"]: p["new_name"]}),
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
# The script player resolves against this: every chooser op plus the internal
|
|
222
|
+
# rename op (header-click renames are logged under the key "rename_single",
|
|
223
|
+
# so saved scripts contain those lines and have to play back).
|
|
224
|
+
PLAY_REGISTRY: dict[str, Operation] = {**OP_REGISTRY, RENAME_OP.key: RENAME_OP}
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
# --- pipeline ---------------------------------------------------------------
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
@dataclass
|
|
231
|
+
class AppliedOp:
|
|
232
|
+
op: Operation
|
|
233
|
+
params: dict[str, Any]
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
@dataclass
|
|
237
|
+
class Pipeline:
|
|
238
|
+
"""Original DataFrame + ordered applied ops; every change re-folds from source."""
|
|
239
|
+
|
|
240
|
+
source: pl.DataFrame
|
|
241
|
+
applied: list[AppliedOp] = field(default_factory=list)
|
|
242
|
+
redo_stack: list[AppliedOp] = field(default_factory=list)
|
|
243
|
+
|
|
244
|
+
def current(self) -> pl.DataFrame:
|
|
245
|
+
df = self.source
|
|
246
|
+
for step in self.applied:
|
|
247
|
+
df = step.op.apply(df, step.params)
|
|
248
|
+
return df
|
|
249
|
+
|
|
250
|
+
def validate(self, op: Operation, params: dict[str, Any]) -> str | None:
|
|
251
|
+
cols = list(self.current().columns)
|
|
252
|
+
for spec in op.params:
|
|
253
|
+
val = params.get(spec.name)
|
|
254
|
+
if spec.kind == "column":
|
|
255
|
+
if not isinstance(val, str) or val not in cols:
|
|
256
|
+
return f"{op.title}: pick a column"
|
|
257
|
+
elif spec.kind == "column_multi":
|
|
258
|
+
picked = [c for c in (val or []) if c in cols]
|
|
259
|
+
if not picked:
|
|
260
|
+
return f"{op.title}: pick at least one column"
|
|
261
|
+
need = 2 if op.key == "combine_columns" else 1
|
|
262
|
+
if len(set(picked)) < need:
|
|
263
|
+
return f"{op.title}: pick at least {need} columns"
|
|
264
|
+
elif spec.kind == "text":
|
|
265
|
+
# split_column's blank delimiter is allowed: it means
|
|
266
|
+
# "split on runs of whitespace"
|
|
267
|
+
if op.key == "combine_columns" and spec.name == "new_name" and not val:
|
|
268
|
+
return f"{op.title}: new name required"
|
|
269
|
+
if (
|
|
270
|
+
op.key == "combine_columns"
|
|
271
|
+
and spec.name == "new_name"
|
|
272
|
+
and val in cols
|
|
273
|
+
):
|
|
274
|
+
return f"{op.title}: {val!r} already exists"
|
|
275
|
+
elif spec.kind == "number":
|
|
276
|
+
if val is None:
|
|
277
|
+
return f"{op.title}: set a value for {spec.title}"
|
|
278
|
+
if spec.min_value is not None and val < spec.min_value:
|
|
279
|
+
return f"{op.title}: {spec.title} must be >= {spec.min_value}"
|
|
280
|
+
return None
|
|
281
|
+
|
|
282
|
+
def apply(self, op: Operation, params: dict[str, Any]) -> None:
|
|
283
|
+
if (err := self.validate(op, params)) is not None:
|
|
284
|
+
raise ValueError(err)
|
|
285
|
+
self.applied.append(AppliedOp(op, dict(params)))
|
|
286
|
+
self.redo_stack.clear()
|
|
287
|
+
|
|
288
|
+
def undo(self) -> bool:
|
|
289
|
+
if not self.applied:
|
|
290
|
+
return False
|
|
291
|
+
self.redo_stack.append(self.applied.pop())
|
|
292
|
+
return True
|
|
293
|
+
|
|
294
|
+
def redo(self) -> bool:
|
|
295
|
+
if not self.redo_stack:
|
|
296
|
+
return False
|
|
297
|
+
self.applied.append(self.redo_stack.pop())
|
|
298
|
+
return True
|
|
299
|
+
|
|
300
|
+
def step_summary(self) -> list[str]:
|
|
301
|
+
out = []
|
|
302
|
+
for step in self.applied:
|
|
303
|
+
bits = " ".join(
|
|
304
|
+
f"{k}={v}" for k, v in step.params.items() if v not in (None, ())
|
|
305
|
+
)
|
|
306
|
+
out.append(f"{step.op.key}({bits})" if bits else step.op.key)
|
|
307
|
+
return out
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
# --- column analysis ---------------------------------------------------------
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
@dataclass
|
|
314
|
+
class ColumnInfo:
|
|
315
|
+
name: str
|
|
316
|
+
dtype: str
|
|
317
|
+
null_count: int
|
|
318
|
+
date_candidate: bool = False
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def analyze_column(df: pl.DataFrame, col: str) -> ColumnInfo:
|
|
322
|
+
"""dtype, null count, and a date-candidate flag from a 200-row sample."""
|
|
323
|
+
s = df.get_column(col)
|
|
324
|
+
info = ColumnInfo(name=col, dtype=str(s.dtype), null_count=int(s.null_count()))
|
|
325
|
+
if s.dtype == pl.String:
|
|
326
|
+
sample = (
|
|
327
|
+
s.drop_nulls().str.strip_chars().replace("", None).drop_nulls().head(200)
|
|
328
|
+
)
|
|
329
|
+
if sample.len() >= 3:
|
|
330
|
+
parsed = date_parser.parse_series(sample)
|
|
331
|
+
ratio = 1.0 - (parsed.null_count() / sample.len())
|
|
332
|
+
info.date_candidate = ratio >= 0.5
|
|
333
|
+
return info
|