normalize-tabular-data 0.1.2__tar.gz → 0.1.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/LICENSE +1 -1
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/PKG-INFO +15 -7
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/README.md +12 -5
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/pyproject.toml +4 -2
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/pyproject.toml.orig +4 -2
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/__init__.py +1 -1
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/app.py +64 -3
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/io.py +21 -3
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/ops.py +44 -27
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/screens.py +9 -8
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/widgets.py +5 -5
- {normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/__main__.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: normalize-tabular-data
|
|
3
|
-
Version: 0.1.
|
|
4
|
-
Summary: TUI for normalizing tabular data with
|
|
3
|
+
Version: 0.1.3
|
|
4
|
+
Summary: TUI for normalizing tabular data with Polars
|
|
5
5
|
Author: David Mertz, Ph.D.
|
|
6
6
|
Author-email: David Mertz, Ph.D. <mertz@gnosis.cx>
|
|
7
7
|
License-Expression: BSD-2-Clause
|
|
@@ -11,6 +11,7 @@ Requires-Dist: textual>=8.2.8
|
|
|
11
11
|
Requires-Dist: polars>=1.44,<2
|
|
12
12
|
Requires-Dist: fastexcel>=0.10
|
|
13
13
|
Requires-Dist: xlsxwriter>=0.9
|
|
14
|
+
Requires-Dist: platformdirs>=3.0
|
|
14
15
|
Requires-Python: >=3.14
|
|
15
16
|
Description-Content-Type: text/markdown
|
|
16
17
|
|
|
@@ -119,17 +120,22 @@ Launch with `normalize-tabular-data`. Keys:
|
|
|
119
120
|
|
|
120
121
|
## Operations
|
|
121
122
|
|
|
122
|
-
- **Normalize
|
|
123
|
-
canonical UTC
|
|
124
|
-
|
|
123
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
124
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
125
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
126
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
127
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
128
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
129
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
130
|
+
(`year-month-day`); unparseable values become null.
|
|
125
131
|
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
126
132
|
per selected columns.
|
|
127
133
|
- Rename a column — click its header in the preview and type the new name.
|
|
128
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
129
134
|
- **Combine columns** — concatenate two or more columns with a separator.
|
|
130
135
|
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
131
136
|
delimiter splits on runs of whitespace.
|
|
132
137
|
- **Remove columns** — drop selected columns entirely.
|
|
138
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
133
139
|
|
|
134
140
|
## Formats
|
|
135
141
|
|
|
@@ -144,7 +150,9 @@ still runs on the full data.
|
|
|
144
150
|
Every open file keeps an internal log of the operations performed on it. When
|
|
145
151
|
saving, the dialog offers a **Save sequence of operations?** checkbox
|
|
146
152
|
(unticked by default); with it ticked, a second dialog asks where to write
|
|
147
|
-
the script — suggested `<table>.ntd
|
|
153
|
+
the script — suggested `<table>.ntd`; scripts always use the `.ntd`
|
|
154
|
+
extension, so the name you type is saved with that suffix whatever it
|
|
155
|
+
ends in.
|
|
148
156
|
|
|
149
157
|
The script is plain ASCII text, one operation per line, e.g.:
|
|
150
158
|
|
|
@@ -103,17 +103,22 @@ Launch with `normalize-tabular-data`. Keys:
|
|
|
103
103
|
|
|
104
104
|
## Operations
|
|
105
105
|
|
|
106
|
-
- **Normalize
|
|
107
|
-
canonical UTC
|
|
108
|
-
|
|
106
|
+
- **Normalize date(t)imes** — parse a messy date column of *any* input
|
|
107
|
+
format into canonical UTC datetimes at millisecond resolution;
|
|
108
|
+
unparseable values become null. CSV/TSV/JSONLines export serializes
|
|
109
|
+
datetime columns as second-resolution `year-month-dayThh:mm:ss`
|
|
110
|
+
strings; Parquet and Excel keep the real datetime values.
|
|
111
|
+
- **Normalize (d)ates** — the same match-anything parsing, condensed to
|
|
112
|
+
the UTC calendar date: the column becomes date-only ISO-8601
|
|
113
|
+
(`year-month-day`); unparseable values become null.
|
|
109
114
|
- **Trim whitespace** — strip edges and collapse internal whitespace runs,
|
|
110
115
|
per selected columns.
|
|
111
116
|
- Rename a column — click its header in the preview and type the new name.
|
|
112
|
-
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
113
117
|
- **Combine columns** — concatenate two or more columns with a separator.
|
|
114
118
|
- **Split column** — break one column into `{col}_1..{col}_k`; a blank
|
|
115
119
|
delimiter splits on runs of whitespace.
|
|
116
120
|
- **Remove columns** — drop selected columns entirely.
|
|
121
|
+
- **Deduplicate rows** — on all or selected columns, keeping first or last.
|
|
117
122
|
|
|
118
123
|
## Formats
|
|
119
124
|
|
|
@@ -128,7 +133,9 @@ still runs on the full data.
|
|
|
128
133
|
Every open file keeps an internal log of the operations performed on it. When
|
|
129
134
|
saving, the dialog offers a **Save sequence of operations?** checkbox
|
|
130
135
|
(unticked by default); with it ticked, a second dialog asks where to write
|
|
131
|
-
the script — suggested `<table>.ntd
|
|
136
|
+
the script — suggested `<table>.ntd`; scripts always use the `.ntd`
|
|
137
|
+
extension, so the name you type is saved with that suffix whatever it
|
|
138
|
+
ends in.
|
|
132
139
|
|
|
133
140
|
The script is plain ASCII text, one operation per line, e.g.:
|
|
134
141
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.1.
|
|
4
|
-
description = "TUI for normalizing tabular data with
|
|
3
|
+
version = "0.1.3"
|
|
4
|
+
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
@@ -12,6 +12,7 @@ dependencies = [
|
|
|
12
12
|
"polars>=1.44,<2",
|
|
13
13
|
"fastexcel>=0.10",
|
|
14
14
|
"xlsxwriter>=0.9",
|
|
15
|
+
"platformdirs>=3.0",
|
|
15
16
|
]
|
|
16
17
|
|
|
17
18
|
[[project.authors]]
|
|
@@ -27,6 +28,7 @@ dev = [
|
|
|
27
28
|
"pytest-asyncio>=0.25",
|
|
28
29
|
"ruff>=0.16.10",
|
|
29
30
|
"twine>=7.0.0",
|
|
31
|
+
"ty>=0.0.84",
|
|
30
32
|
]
|
|
31
33
|
|
|
32
34
|
[build-system]
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "normalize-tabular-data"
|
|
3
|
-
version = "0.1.
|
|
4
|
-
description = "TUI for normalizing tabular data with
|
|
3
|
+
version = "0.1.3"
|
|
4
|
+
description = "TUI for normalizing tabular data with Polars"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "BSD-2-Clause"
|
|
7
7
|
license-files = ["LICENSE"]
|
|
@@ -15,6 +15,7 @@ dependencies = [
|
|
|
15
15
|
"polars>=1.44,<2",
|
|
16
16
|
"fastexcel>=0.10",
|
|
17
17
|
"xlsxwriter>=0.9",
|
|
18
|
+
"platformdirs>=3.0",
|
|
18
19
|
]
|
|
19
20
|
|
|
20
21
|
[dependency-groups]
|
|
@@ -23,6 +24,7 @@ dev = [
|
|
|
23
24
|
"pytest-asyncio>=0.25",
|
|
24
25
|
"ruff>=0.16.10",
|
|
25
26
|
"twine>=7.0.0",
|
|
27
|
+
"ty>=0.0.84",
|
|
26
28
|
]
|
|
27
29
|
|
|
28
30
|
[project.scripts]
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/app.py
RENAMED
|
@@ -6,6 +6,7 @@ import bisect
|
|
|
6
6
|
import datetime as _dt
|
|
7
7
|
import random
|
|
8
8
|
from pathlib import Path
|
|
9
|
+
from platformdirs import user_config_dir
|
|
9
10
|
from typing import Any, Iterable
|
|
10
11
|
|
|
11
12
|
from textual import on
|
|
@@ -43,6 +44,7 @@ from normalize_tabular_data.screens import (
|
|
|
43
44
|
from normalize_tabular_data.widgets import ColumnSidebar, MenuFooter, StepsBar
|
|
44
45
|
|
|
45
46
|
PREVIEW_ROWS = 250
|
|
47
|
+
APP_CONFIG_NAME = "normalize-tabular-data"
|
|
46
48
|
|
|
47
49
|
|
|
48
50
|
class CommandMenu(CommandPalette):
|
|
@@ -135,10 +137,19 @@ class OpChooserModal(ModalDialog):
|
|
|
135
137
|
|
|
136
138
|
def _marked_title(self, op_key: str, title: str, hotkey: str = "") -> str:
|
|
137
139
|
"""Return "Normalize (d)ates"-style markup: the op's hotkey letter
|
|
138
|
-
highlighted where it sits in the title (keeping its case).
|
|
139
|
-
|
|
140
|
-
|
|
140
|
+
highlighted where it sits in the title (keeping its case). A title
|
|
141
|
+
that already parenthesizes its letter ("Normalize date(t)imes")
|
|
142
|
+
highlights the letter between the parens; a plain title matches
|
|
143
|
+
its first occurrence and adds the parens. Ops without a designated
|
|
144
|
+
hotkey claim the first unused letter in the title instead; plain
|
|
145
|
+
title if no letter can be claimed."""
|
|
141
146
|
if hotkey:
|
|
147
|
+
i = title.lower().find(f"({hotkey.lower()})")
|
|
148
|
+
if i >= 0:
|
|
149
|
+
# the letter to claim sits inside existing parentheses:
|
|
150
|
+
# highlight it without adding a second pair
|
|
151
|
+
self._hotkeys[title[i + 1].lower()] = op_key
|
|
152
|
+
return f"{title[: i + 1]}[cyan]{title[i + 1]}[/cyan]{title[i + 2:]}"
|
|
142
153
|
i = title.lower().find(hotkey.lower())
|
|
143
154
|
else:
|
|
144
155
|
i = next(
|
|
@@ -275,8 +286,15 @@ class NormalizeApp(App[None]):
|
|
|
275
286
|
self,
|
|
276
287
|
initial_file: str | Path | None = None,
|
|
277
288
|
initial_script: str | Path | None = None,
|
|
289
|
+
config_dir: Path | None = None,
|
|
278
290
|
) -> None:
|
|
279
291
|
super().__init__()
|
|
292
|
+
# where the chosen theme (and any future settings) are stored;
|
|
293
|
+
# None means the per-user platform directory
|
|
294
|
+
self._config_dir = config_dir
|
|
295
|
+
# flips to True at on_mount: only theme changes made after are the
|
|
296
|
+
# user's and only those are persisted
|
|
297
|
+
self._app_ready = False
|
|
280
298
|
# filesystem path to open when the app starts (the optional file
|
|
281
299
|
# named on the command line); loaded like a file chosen in-app
|
|
282
300
|
self.initial_file: Path | None = (
|
|
@@ -354,6 +372,9 @@ class NormalizeApp(App[None]):
|
|
|
354
372
|
return list(self.pipeline.current().columns)
|
|
355
373
|
|
|
356
374
|
def on_mount(self) -> None:
|
|
375
|
+
# remember this point so theme changes from here on are the
|
|
376
|
+
# user's own choices and get persisted
|
|
377
|
+
self._apply_saved_theme()
|
|
357
378
|
self.push_screen(MainScreen())
|
|
358
379
|
self._refresh_steps()
|
|
359
380
|
# the command line's file, if any, is opened once the main screen
|
|
@@ -361,6 +382,46 @@ class NormalizeApp(App[None]):
|
|
|
361
382
|
# crashing if it is unreadable or has an unsupported format
|
|
362
383
|
if self.initial_file is not None:
|
|
363
384
|
self.call_after_refresh(self.load_path, self.initial_file)
|
|
385
|
+
self._app_ready = True
|
|
386
|
+
|
|
387
|
+
# --- theme persistence ---------------------------------------------
|
|
388
|
+
|
|
389
|
+
def watch_theme(self, theme_name: str) -> None:
|
|
390
|
+
"""Persist every theme the user chooses (menu or command), so the
|
|
391
|
+
next launch opens with it again. Skipped while headless, so the
|
|
392
|
+
pilot tests cannot touch the real configuration; skipped before
|
|
393
|
+
mount, where only the reactive's default applies."""
|
|
394
|
+
if self._app_ready and not self.is_headless:
|
|
395
|
+
try:
|
|
396
|
+
self._theme_config_path().parent.mkdir(parents=True, exist_ok=True)
|
|
397
|
+
self._theme_config_path().write_text(
|
|
398
|
+
theme_name + "\n", encoding="utf-8"
|
|
399
|
+
)
|
|
400
|
+
except OSError:
|
|
401
|
+
pass # unwritable config location: run without remembering
|
|
402
|
+
|
|
403
|
+
def _apply_saved_theme(self) -> None:
|
|
404
|
+
"""Restore the theme chosen in a previous session, if it still exists.
|
|
405
|
+
|
|
406
|
+
Skipped in headless runs that use the real configuration location
|
|
407
|
+
(pilot tests): an eyeballing developer's own saved theme should
|
|
408
|
+
not leak into supposedly default-themed tests. A headless run
|
|
409
|
+
given an explicit config_dir is a deliberate fixture and loads it."""
|
|
410
|
+
if self.is_headless and self._config_dir is None:
|
|
411
|
+
return
|
|
412
|
+
try:
|
|
413
|
+
saved = self._theme_config_path().read_text(encoding="utf-8").strip()
|
|
414
|
+
except OSError:
|
|
415
|
+
return
|
|
416
|
+
if not saved:
|
|
417
|
+
return
|
|
418
|
+
try:
|
|
419
|
+
self.theme = saved
|
|
420
|
+
except Exception:
|
|
421
|
+
pass # a theme name that no longer exists: keep the default
|
|
422
|
+
|
|
423
|
+
def _theme_config_path(self) -> Path:
|
|
424
|
+
return (self._config_dir or Path(user_config_dir(APP_CONFIG_NAME))) / "theme"
|
|
364
425
|
|
|
365
426
|
# --- data plumbing ---------------------------------------------------
|
|
366
427
|
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/io.py
RENAMED
|
@@ -79,6 +79,24 @@ def excel_sheets(path: Path) -> list[str]:
|
|
|
79
79
|
return list(fastexcel.read_excel(path).sheet_names)
|
|
80
80
|
|
|
81
81
|
|
|
82
|
+
def _second_resolution(df: pl.DataFrame) -> pl.DataFrame:
|
|
83
|
+
"""Datetime columns serialized as canonical UTC second-resolution text.
|
|
84
|
+
|
|
85
|
+
Used only for the text formats (CSV/TSV/JSONL): polars pads datetime
|
|
86
|
+
output with sub-second digits no matter the unit, and the canonical
|
|
87
|
+
text form carries no fractional part. Parquet and Excel keep the real
|
|
88
|
+
Datetime values (millisecond resolution)."""
|
|
89
|
+
datetime_cols = [
|
|
90
|
+
c for c, dtype in df.schema.items() if isinstance(dtype, pl.Datetime)
|
|
91
|
+
]
|
|
92
|
+
if not datetime_cols:
|
|
93
|
+
return df
|
|
94
|
+
return df.with_columns(
|
|
95
|
+
pl.col(c).dt.truncate("1s").dt.to_string("%Y-%m-%dT%H:%M:%S")
|
|
96
|
+
for c in datetime_cols
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
82
100
|
def write_table(
|
|
83
101
|
df: pl.DataFrame,
|
|
84
102
|
path: Path,
|
|
@@ -86,11 +104,11 @@ def write_table(
|
|
|
86
104
|
sheet_name: str = "data",
|
|
87
105
|
) -> None:
|
|
88
106
|
if fmt == "csv":
|
|
89
|
-
df.write_csv(path)
|
|
107
|
+
_second_resolution(df).write_csv(path)
|
|
90
108
|
elif fmt == "tsv":
|
|
91
|
-
df.write_csv(path, separator="\t")
|
|
109
|
+
_second_resolution(df).write_csv(path, separator="\t")
|
|
92
110
|
elif fmt == "jsonl":
|
|
93
|
-
df.write_ndjson(path)
|
|
111
|
+
_second_resolution(df).write_ndjson(path)
|
|
94
112
|
elif fmt == "parquet":
|
|
95
113
|
df.write_parquet(path)
|
|
96
114
|
elif fmt == "xlsx":
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/ops.py
RENAMED
|
@@ -48,19 +48,22 @@ def _apply_date_normalize(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
|
48
48
|
series = df.get_column(col)
|
|
49
49
|
if series.dtype != pl.String:
|
|
50
50
|
series = series.cast(pl.String)
|
|
51
|
-
# canonical UTC at
|
|
52
|
-
# Datetime (ns/us/ms only, all of which print sub-second digits in
|
|
53
|
-
# every export), so the normalized column is the truncated
|
|
54
|
-
# `year-month-dayThh:mm:ss` string; fractional parts are dropped and
|
|
55
|
-
# unparseable values become null
|
|
51
|
+
# canonical UTC Datetime at millisecond resolution; unparseable -> null
|
|
56
52
|
return df.with_columns(
|
|
57
|
-
date_parser.parse_series(series)
|
|
58
|
-
.dt.truncate("1s")
|
|
59
|
-
.dt.to_string("%Y-%m-%dT%H:%M:%S")
|
|
60
|
-
.alias(col)
|
|
53
|
+
date_parser.parse_series(series).dt.cast_time_unit("ms").alias(col)
|
|
61
54
|
)
|
|
62
55
|
|
|
63
56
|
|
|
57
|
+
def _apply_date_only(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
58
|
+
col: str = p["column"]
|
|
59
|
+
series = df.get_column(col)
|
|
60
|
+
if series.dtype != pl.String:
|
|
61
|
+
series = series.cast(pl.String)
|
|
62
|
+
# date-only ISO-8601: any input format condensed to the UTC calendar
|
|
63
|
+
# date; unparseable values become null
|
|
64
|
+
return df.with_columns(date_parser.parse_series(series).dt.date().alias(col))
|
|
65
|
+
|
|
66
|
+
|
|
64
67
|
def _apply_trim_collapse(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
65
68
|
cols: list[str] = p["columns"]
|
|
66
69
|
return df.with_columns(
|
|
@@ -126,8 +129,8 @@ def _apply_split_column(df: pl.DataFrame, p: dict[str, Any]) -> pl.DataFrame:
|
|
|
126
129
|
OPS: tuple[Operation, ...] = (
|
|
127
130
|
Operation(
|
|
128
131
|
key="date_normalize",
|
|
129
|
-
title="Normalize
|
|
130
|
-
hotkey="
|
|
132
|
+
title="Normalize date(t)imes",
|
|
133
|
+
hotkey="t",
|
|
131
134
|
params=(
|
|
132
135
|
ParamSpec(
|
|
133
136
|
"column",
|
|
@@ -138,6 +141,20 @@ OPS: tuple[Operation, ...] = (
|
|
|
138
141
|
),
|
|
139
142
|
apply=_apply_date_normalize,
|
|
140
143
|
),
|
|
144
|
+
Operation(
|
|
145
|
+
key="date_only",
|
|
146
|
+
title="Normalize (d)ates",
|
|
147
|
+
hotkey="d",
|
|
148
|
+
params=(
|
|
149
|
+
ParamSpec(
|
|
150
|
+
"column",
|
|
151
|
+
"column",
|
|
152
|
+
"Date column",
|
|
153
|
+
help="Any input format; condenses to the UTC calendar date",
|
|
154
|
+
),
|
|
155
|
+
),
|
|
156
|
+
apply=_apply_date_only,
|
|
157
|
+
),
|
|
141
158
|
Operation(
|
|
142
159
|
key="trim_collapse",
|
|
143
160
|
title="Trim whitespace",
|
|
@@ -152,22 +169,6 @@ OPS: tuple[Operation, ...] = (
|
|
|
152
169
|
),
|
|
153
170
|
apply=_apply_trim_collapse,
|
|
154
171
|
),
|
|
155
|
-
Operation(
|
|
156
|
-
key="dedup_rows",
|
|
157
|
-
title="Deduplicate rows",
|
|
158
|
-
hotkey="p",
|
|
159
|
-
params=(
|
|
160
|
-
ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
|
|
161
|
-
ParamSpec(
|
|
162
|
-
"keep",
|
|
163
|
-
"choice",
|
|
164
|
-
"Keep",
|
|
165
|
-
default="first",
|
|
166
|
-
choices=("first", "last"),
|
|
167
|
-
),
|
|
168
|
-
),
|
|
169
|
-
apply=_apply_dedup_rows,
|
|
170
|
-
),
|
|
171
172
|
Operation(
|
|
172
173
|
key="combine_columns",
|
|
173
174
|
title="Combine columns",
|
|
@@ -212,6 +213,22 @@ OPS: tuple[Operation, ...] = (
|
|
|
212
213
|
),
|
|
213
214
|
apply=_apply_drop_columns,
|
|
214
215
|
),
|
|
216
|
+
Operation(
|
|
217
|
+
key="dedup_rows",
|
|
218
|
+
title="Deduplicate rows",
|
|
219
|
+
hotkey="p",
|
|
220
|
+
params=(
|
|
221
|
+
ParamSpec("columns", "column_multi", "Key columns (empty = all)"),
|
|
222
|
+
ParamSpec(
|
|
223
|
+
"keep",
|
|
224
|
+
"choice",
|
|
225
|
+
"Keep",
|
|
226
|
+
default="first",
|
|
227
|
+
choices=("first", "last"),
|
|
228
|
+
),
|
|
229
|
+
),
|
|
230
|
+
apply=_apply_dedup_rows,
|
|
231
|
+
),
|
|
215
232
|
)
|
|
216
233
|
|
|
217
234
|
OP_REGISTRY: dict[str, Operation] = {op.key: op for op in OPS}
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/screens.py
RENAMED
|
@@ -363,7 +363,11 @@ class OpParamsModal(ModalDialog):
|
|
|
363
363
|
|
|
364
364
|
def compose_body(self) -> ComposeResult:
|
|
365
365
|
self.widgets: dict[str, object] = {}
|
|
366
|
-
candidates =
|
|
366
|
+
candidates = (
|
|
367
|
+
self._date_candidates()
|
|
368
|
+
if self.op.key in ("date_normalize", "date_only")
|
|
369
|
+
else []
|
|
370
|
+
)
|
|
367
371
|
for spec in self.op.params:
|
|
368
372
|
if spec.kind == "column_multi":
|
|
369
373
|
sel = SelectionList(
|
|
@@ -531,8 +535,8 @@ class SaveModal(ModalDialog):
|
|
|
531
535
|
class ScriptNameModal(ModalDialog):
|
|
532
536
|
"""Name the '.ntd' script file when saving a sequence of operations.
|
|
533
537
|
|
|
534
|
-
The
|
|
535
|
-
|
|
538
|
+
The extension is always .ntd: whatever the user types, the saved name
|
|
539
|
+
ends in the fixed script suffix."""
|
|
536
540
|
|
|
537
541
|
dialog_title = "Save script"
|
|
538
542
|
|
|
@@ -547,17 +551,14 @@ class ScriptNameModal(ModalDialog):
|
|
|
547
551
|
id="script_input",
|
|
548
552
|
)
|
|
549
553
|
yield self.path_input
|
|
550
|
-
yield Static(
|
|
551
|
-
f"Default extension: {SCRIPT_SUFFIX} — type another to override",
|
|
552
|
-
classes="help",
|
|
553
|
-
)
|
|
554
|
+
yield Static(f"Scripts always end in {SCRIPT_SUFFIX}", classes="help")
|
|
554
555
|
|
|
555
556
|
def action_ok(self) -> None:
|
|
556
557
|
raw = self.path_input.value.strip()
|
|
557
558
|
if not raw:
|
|
558
559
|
self.app.notify("Type a script file name or path", severity="error")
|
|
559
560
|
return
|
|
560
|
-
target = Path(raw).expanduser()
|
|
561
|
+
target = Path(raw).expanduser().with_suffix(SCRIPT_SUFFIX)
|
|
561
562
|
if target.exists():
|
|
562
563
|
self.app.notify("Script file exists: pick another name", severity="warning")
|
|
563
564
|
return
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/widgets.py
RENAMED
|
@@ -92,13 +92,13 @@ class ColumnSidebar(Widget):
|
|
|
92
92
|
|
|
93
93
|
|
|
94
94
|
def _dtype_mark(dtype: str) -> str:
|
|
95
|
-
"""Compact 3-wide dtype mark for the sidebar: dt (datetime
|
|
96
|
-
str (string), int (integer), dec (float/decimal),
|
|
97
|
-
everything else oth. Each mark is padded to 3 columns
|
|
98
|
-
names left-align."""
|
|
95
|
+
"""Compact 3-wide dtype mark for the sidebar: dt (datetime),
|
|
96
|
+
day (date-only), str (string), int (integer), dec (float/decimal),
|
|
97
|
+
t/f (boolean); everything else oth. Each mark is padded to 3 columns
|
|
98
|
+
so column names left-align."""
|
|
99
99
|
for prefix, mark in (
|
|
100
100
|
("Datetime", "dt "),
|
|
101
|
-
("Date", "
|
|
101
|
+
("Date", "day"),
|
|
102
102
|
("Time", "dt "),
|
|
103
103
|
("String", "str"),
|
|
104
104
|
("Categorical", "str"),
|
{normalize_tabular_data-0.1.2 → normalize_tabular_data-0.1.3}/src/normalize_tabular_data/__main__.py
RENAMED
|
File without changes
|