eosframes 1.1.0__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eosframes-1.1.0 → eosframes-1.2.0}/PKG-INFO +33 -2
- {eosframes-1.1.0 → eosframes-1.2.0}/README.md +31 -1
- {eosframes-1.1.0 → eosframes-1.2.0}/pyproject.toml +1 -1
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/cli.py +49 -6
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/ops.py +1 -3
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/scale.py +487 -121
- eosframes-1.2.0/src/eosframes/utils.py +126 -0
- eosframes-1.1.0/src/eosframes/utils.py +0 -23
- {eosframes-1.1.0 → eosframes-1.2.0}/LICENSE +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/__init__.py +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/exceptions.py +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/hub.py +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/logger.py +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/naming.py +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/read.py +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/stack.py +0 -0
- {eosframes-1.1.0 → eosframes-1.2.0}/src/eosframes/write.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: eosframes
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Ersilia utilities for working with tabular output data
|
|
5
5
|
License: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -19,6 +19,7 @@ Classifier: Programming Language :: Python :: 3.11
|
|
|
19
19
|
Classifier: Programming Language :: Python :: 3.12
|
|
20
20
|
Classifier: Programming Language :: Python :: 3.13
|
|
21
21
|
Classifier: Programming Language :: Python :: 3.14
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
22
23
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
24
|
Classifier: Topic :: Scientific/Engineering :: Chemistry
|
|
24
25
|
Requires-Dist: click (>=8.0)
|
|
@@ -98,11 +99,41 @@ Run `eosframes --help` (or `eosframes <command> --help`) for inline help.
|
|
|
98
99
|
|
|
99
100
|
See [`docs/cli.md`](docs/cli.md) for every flag, example, and refusal condition.
|
|
100
101
|
|
|
102
|
+
`fit` and `transform` stream in bounded memory — `fit` walks one column at a
|
|
103
|
+
time, `transform` one row-chunk at a time — so they handle files far larger
|
|
104
|
+
than RAM (wide fingerprint frames, tens of GB). Tune `--chunksize` to trade
|
|
105
|
+
memory for throughput.
|
|
106
|
+
|
|
107
|
+
## Scripts
|
|
108
|
+
|
|
109
|
+
[`scripts/build_scaler.sh`](scripts/build_scaler.sh) builds and packages a scaler
|
|
110
|
+
for one model in a single step: it pulls the model's precalculations from the
|
|
111
|
+
[isaura](https://github.com/ersilia-os/isaura) store over the Ersilia reference
|
|
112
|
+
library (`data/ersilia_reference_library_v0.csv`), fits an `eosframes` scaler, and
|
|
113
|
+
compresses the transformer into a versioned zip.
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
scripts/build_scaler.sh <model_id> <version> # e.g. eos4e40 v1
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
The artifact is written to:
|
|
120
|
+
|
|
121
|
+
```
|
|
122
|
+
output/ersilia_reference_library_v0/<model_id>/<version>/scaler-<eosframes-major>.zip
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
(e.g. `output/ersilia_reference_library_v0/eos4e40/v1/scaler-1.zip`), containing a
|
|
126
|
+
single `<model_id>_<version>_transformer.json`. The bucket defaults to
|
|
127
|
+
`isaura-public`; override it with the `PROJECT_NAME` environment variable.
|
|
128
|
+
|
|
129
|
+
**Prerequisites:** `isaura` and `eosframes` on `PATH`, and a running local isaura
|
|
130
|
+
MinIO engine (`isaura engine --start`).
|
|
131
|
+
|
|
101
132
|
## Documentation
|
|
102
133
|
|
|
103
134
|
- [`docs/cli.md`](docs/cli.md) — every CLI command, all flags, examples, and error patterns.
|
|
104
135
|
- [`docs/nomenclature.md`](docs/nomenclature.md) — every recognised filename / directory pattern, the strict/lenient contract, and the two stack modes.
|
|
105
|
-
- [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, and
|
|
136
|
+
- [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, quantization / imputation, and the bounded-memory streaming fit/transform for large files.
|
|
106
137
|
|
|
107
138
|
## About the Ersilia Open Source Initiative
|
|
108
139
|
|
|
@@ -65,11 +65,41 @@ Run `eosframes --help` (or `eosframes <command> --help`) for inline help.
|
|
|
65
65
|
|
|
66
66
|
See [`docs/cli.md`](docs/cli.md) for every flag, example, and refusal condition.
|
|
67
67
|
|
|
68
|
+
`fit` and `transform` stream in bounded memory — `fit` walks one column at a
|
|
69
|
+
time, `transform` one row-chunk at a time — so they handle files far larger
|
|
70
|
+
than RAM (wide fingerprint frames, tens of GB). Tune `--chunksize` to trade
|
|
71
|
+
memory for throughput.
|
|
72
|
+
|
|
73
|
+
## Scripts
|
|
74
|
+
|
|
75
|
+
[`scripts/build_scaler.sh`](scripts/build_scaler.sh) builds and packages a scaler
|
|
76
|
+
for one model in a single step: it pulls the model's precalculations from the
|
|
77
|
+
[isaura](https://github.com/ersilia-os/isaura) store over the Ersilia reference
|
|
78
|
+
library (`data/ersilia_reference_library_v0.csv`), fits an `eosframes` scaler, and
|
|
79
|
+
compresses the transformer into a versioned zip.
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
scripts/build_scaler.sh <model_id> <version> # e.g. eos4e40 v1
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
The artifact is written to:
|
|
86
|
+
|
|
87
|
+
```
|
|
88
|
+
output/ersilia_reference_library_v0/<model_id>/<version>/scaler-<eosframes-major>.zip
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
(e.g. `output/ersilia_reference_library_v0/eos4e40/v1/scaler-1.zip`), containing a
|
|
92
|
+
single `<model_id>_<version>_transformer.json`. The bucket defaults to
|
|
93
|
+
`isaura-public`; override it with the `PROJECT_NAME` environment variable.
|
|
94
|
+
|
|
95
|
+
**Prerequisites:** `isaura` and `eosframes` on `PATH`, and a running local isaura
|
|
96
|
+
MinIO engine (`isaura engine --start`).
|
|
97
|
+
|
|
68
98
|
## Documentation
|
|
69
99
|
|
|
70
100
|
- [`docs/cli.md`](docs/cli.md) — every CLI command, all flags, examples, and error patterns.
|
|
71
101
|
- [`docs/nomenclature.md`](docs/nomenclature.md) — every recognised filename / directory pattern, the strict/lenient contract, and the two stack modes.
|
|
72
|
-
- [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, and
|
|
102
|
+
- [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, quantization / imputation, and the bounded-memory streaming fit/transform for large files.
|
|
73
103
|
|
|
74
104
|
## About the Ersilia Open Source Initiative
|
|
75
105
|
|
|
@@ -186,9 +186,7 @@ def _render_sidecar(
|
|
|
186
186
|
click.echo(resolved_out)
|
|
187
187
|
|
|
188
188
|
|
|
189
|
-
def _fetch_or_clickfail(
|
|
190
|
-
fetch: Callable, *args, **kwargs
|
|
191
|
-
):
|
|
189
|
+
def _fetch_or_clickfail(fetch: Callable, *args, **kwargs):
|
|
192
190
|
"""Call a hub fetcher and convert ``EosframesError`` to ``ClickException``."""
|
|
193
191
|
try:
|
|
194
192
|
return fetch(*args, **kwargs)
|
|
@@ -664,7 +662,27 @@ def columns(input_file: str, output: str) -> None:
|
|
|
664
662
|
"transform time."
|
|
665
663
|
),
|
|
666
664
|
)
|
|
667
|
-
|
|
665
|
+
@click.option(
|
|
666
|
+
"--chunksize",
|
|
667
|
+
"chunksize",
|
|
668
|
+
type=int,
|
|
669
|
+
default=_scale._DEFAULT_CHUNKSIZE,
|
|
670
|
+
show_default=True,
|
|
671
|
+
help=(
|
|
672
|
+
"Rows per streamed chunk. The fit never holds the whole file — it "
|
|
673
|
+
"walks one column at a time (CSV inputs are first staged to a "
|
|
674
|
+
"temporary columnar H5 in chunks of this size). Lower it for very "
|
|
675
|
+
"wide frames, raise it for narrow ones."
|
|
676
|
+
),
|
|
677
|
+
)
|
|
678
|
+
def fit(
|
|
679
|
+
input_file: str,
|
|
680
|
+
scaler: str,
|
|
681
|
+
output: str,
|
|
682
|
+
quantize: bool,
|
|
683
|
+
impute: bool,
|
|
684
|
+
chunksize: int,
|
|
685
|
+
) -> None:
|
|
668
686
|
"""Fit a type-aware robust scaler on INPUT_FILE and save parameters to SCALER.
|
|
669
687
|
|
|
670
688
|
Each numeric feature column is auto-classified (constant, binary, count,
|
|
@@ -700,6 +718,7 @@ def fit(input_file: str, scaler: str, output: str, quantize: bool, impute: bool)
|
|
|
700
718
|
output_path=output,
|
|
701
719
|
output_dtype=output_dtype,
|
|
702
720
|
impute=impute,
|
|
721
|
+
chunksize=chunksize,
|
|
703
722
|
)
|
|
704
723
|
except EosframesError as e:
|
|
705
724
|
raise _err(e) from e
|
|
@@ -744,7 +763,26 @@ def fit(input_file: str, scaler: str, output: str, quantize: bool, impute: bool)
|
|
|
744
763
|
"no -128 sentinels)."
|
|
745
764
|
),
|
|
746
765
|
)
|
|
747
|
-
|
|
766
|
+
@click.option(
|
|
767
|
+
"--chunksize",
|
|
768
|
+
"chunksize",
|
|
769
|
+
type=int,
|
|
770
|
+
default=_scale._DEFAULT_CHUNKSIZE,
|
|
771
|
+
show_default=True,
|
|
772
|
+
help=(
|
|
773
|
+
"Rows per streamed chunk. The input is read and the output written "
|
|
774
|
+
"one chunk at a time, so memory stays bounded regardless of file "
|
|
775
|
+
"size. Lower it for very wide frames, raise it for narrow ones."
|
|
776
|
+
),
|
|
777
|
+
)
|
|
778
|
+
def transform(
|
|
779
|
+
input_file: str,
|
|
780
|
+
scaler: str,
|
|
781
|
+
output: str,
|
|
782
|
+
quantize: bool,
|
|
783
|
+
impute: bool,
|
|
784
|
+
chunksize: int,
|
|
785
|
+
) -> None:
|
|
748
786
|
"""Apply a saved scaler to INPUT_FILE and write scaled data to OUTPUT.
|
|
749
787
|
|
|
750
788
|
Loads the scaler parameters from SCALER and applies them to the numeric
|
|
@@ -757,7 +795,12 @@ def transform(input_file: str, scaler: str, output: str, quantize: bool, impute:
|
|
|
757
795
|
output_dtype = "int8" if quantize else "float32"
|
|
758
796
|
try:
|
|
759
797
|
out = _scale.transform_file(
|
|
760
|
-
input_file,
|
|
798
|
+
input_file,
|
|
799
|
+
scaler,
|
|
800
|
+
output,
|
|
801
|
+
output_dtype=output_dtype,
|
|
802
|
+
impute=impute,
|
|
803
|
+
chunksize=chunksize,
|
|
761
804
|
)
|
|
762
805
|
except EosframesError as e:
|
|
763
806
|
raise _err(e) from e
|
|
@@ -76,9 +76,7 @@ def _require_no_overwrite(path: str, *, kind: str = "file") -> None:
|
|
|
76
76
|
f"Output folder '{path}' already exists. "
|
|
77
77
|
"Remove it or choose a different name."
|
|
78
78
|
)
|
|
79
|
-
raise EosframesError(
|
|
80
|
-
f"Output file '{path}' already exists. Remove it first."
|
|
81
|
-
)
|
|
79
|
+
raise EosframesError(f"Output file '{path}' already exists. Remove it first.")
|
|
82
80
|
|
|
83
81
|
|
|
84
82
|
def _compute_summary_stats(df: pd.DataFrame) -> List[Dict]:
|
|
@@ -173,8 +173,9 @@ hand-maintained schema number.
|
|
|
173
173
|
|
|
174
174
|
import json
|
|
175
175
|
import os
|
|
176
|
+
import tempfile
|
|
176
177
|
from datetime import datetime
|
|
177
|
-
from typing import Callable, Dict, Optional, Tuple
|
|
178
|
+
from typing import Callable, Dict, Iterator, Optional, Tuple
|
|
178
179
|
|
|
179
180
|
import h5py
|
|
180
181
|
import numpy as np
|
|
@@ -189,6 +190,7 @@ from .naming import (
|
|
|
189
190
|
parse_name,
|
|
190
191
|
parse_transformer_name,
|
|
191
192
|
)
|
|
193
|
+
from .utils import progress
|
|
192
194
|
|
|
193
195
|
_META_COLS = {"key", "input"}
|
|
194
196
|
|
|
@@ -197,6 +199,31 @@ _METHOD_NAME = "robust_typed"
|
|
|
197
199
|
_VALID_OUTPUT_DTYPES = ("float32", "int8")
|
|
198
200
|
_DEFAULT_OUTPUT_DTYPE = "float32"
|
|
199
201
|
|
|
202
|
+
# Rows per chunk for the streaming file-level fit/transform paths. Big
|
|
203
|
+
# enough to amortise per-chunk overhead, small enough that one chunk of a
|
|
204
|
+
# wide frame stays well under a GB (e.g. 50k rows × 3k cols × 4B ≈ 0.6 GB).
|
|
205
|
+
_DEFAULT_CHUNKSIZE = 50_000
|
|
206
|
+
|
|
207
|
+
# Row cadence for the CSV→H5 staging milestone log (independent of chunksize,
|
|
208
|
+
# so the log reads the same whatever chunk size the caller picks).
|
|
209
|
+
_STAGE_LOG_ROWS = 1_000_000
|
|
210
|
+
|
|
211
|
+
# Columns pulled per HDF5 read during the fit. Reading the values dataset one
|
|
212
|
+
# column at a time forces a re-read of every storage chunk the column touches,
|
|
213
|
+
# so we read a batch of columns at once and then fit each column from the
|
|
214
|
+
# in-memory block. 100 cols × n_rows × 4 B stays bounded (~0.5 GB at 1.3 M
|
|
215
|
+
# rows) while cutting the number of HDF5 reads ~100×.
|
|
216
|
+
_FIT_COLUMN_BATCH = 100
|
|
217
|
+
|
|
218
|
+
# Target bytes per HDF5 storage chunk for the staged values dataset. Chunks are
|
|
219
|
+
# tiled ``(row_tile, _FIT_COLUMN_BATCH)`` so the fit's column-batch reads align
|
|
220
|
+
# to chunk boundaries — each chunk is read exactly once across the whole fit.
|
|
221
|
+
_STAGE_CHUNK_BYTES = 4 * 1024 * 1024
|
|
222
|
+
|
|
223
|
+
# Target number of progress log lines for the column-fit and row-chunk
|
|
224
|
+
# transform loops when running off-TTY (one line per ~5% of the work).
|
|
225
|
+
_PROGRESS_LOG_STEPS = 20
|
|
226
|
+
|
|
200
227
|
_INT8_NAN_SENTINEL = -128
|
|
201
228
|
_INT8_MAX_VAL = 127
|
|
202
229
|
|
|
@@ -309,9 +336,7 @@ _COUNT_SHIFTED_EXTENT_RATIO_MAX = 2.0
|
|
|
309
336
|
_PIECEWISE_BLEND_FRACTION = 0.5
|
|
310
337
|
|
|
311
338
|
|
|
312
|
-
def _tanh_tail(
|
|
313
|
-
mag: np.ndarray, body_extent: float, body_target: float
|
|
314
|
-
) -> np.ndarray:
|
|
339
|
+
def _tanh_tail(mag: np.ndarray, body_extent: float, body_target: float) -> np.ndarray:
|
|
315
340
|
"""Asymptotic tail for ``|x - center| > body_extent``.
|
|
316
341
|
|
|
317
342
|
Returns ``y = body_target + (1 − body_target)·tanh(c·u)`` for
|
|
@@ -596,12 +621,10 @@ def _fit_continuous(series: pd.Series) -> dict:
|
|
|
596
621
|
tail_asymmetry_right = right_span / max(left_span, _eps)
|
|
597
622
|
tail_asymmetry_left = left_span / max(right_span, _eps)
|
|
598
623
|
is_right_skew = (
|
|
599
|
-
bowley > _BOWLEY_THRESHOLD
|
|
600
|
-
and tail_asymmetry_right > _TAIL_ASYMMETRY_THRESHOLD
|
|
624
|
+
bowley > _BOWLEY_THRESHOLD and tail_asymmetry_right > _TAIL_ASYMMETRY_THRESHOLD
|
|
601
625
|
)
|
|
602
626
|
is_left_skew = (
|
|
603
|
-
bowley < -_BOWLEY_THRESHOLD
|
|
604
|
-
and tail_asymmetry_left > _TAIL_ASYMMETRY_THRESHOLD
|
|
627
|
+
bowley < -_BOWLEY_THRESHOLD and tail_asymmetry_left > _TAIL_ASYMMETRY_THRESHOLD
|
|
605
628
|
)
|
|
606
629
|
|
|
607
630
|
if is_right_skew:
|
|
@@ -756,6 +779,9 @@ def _fit_count(series: pd.Series) -> dict:
|
|
|
756
779
|
# Flag near-degenerate sparse counts where most rows are 0 and
|
|
757
780
|
# the scaler can only produce a handful of distinct values. Goes
|
|
758
781
|
# into fit_notes — advisory only, never read by the transform.
|
|
782
|
+
# No per-column log here: sparse fingerprints (e.g. Morgan counts)
|
|
783
|
+
# flag hundreds of columns and would flood the log. fit() emits a
|
|
784
|
+
# single aggregated summary instead.
|
|
759
785
|
scaled = np.clip(arr / high_anchor, 0.0, 1.0)
|
|
760
786
|
n_distinct = int(np.unique(scaled).size)
|
|
761
787
|
mode_fraction = float((arr == 0.0).sum()) / float(arr.size)
|
|
@@ -768,15 +794,6 @@ def _fit_count(series: pd.Series) -> dict:
|
|
|
768
794
|
"mode_fraction": mode_fraction,
|
|
769
795
|
"n_distinct_output": n_distinct,
|
|
770
796
|
}
|
|
771
|
-
get_logger().warning(
|
|
772
|
-
"Count column '%s' is near-degenerate "
|
|
773
|
-
"(mode_fraction=%.2f, n_distinct_output=%d). "
|
|
774
|
-
"Output collapses to a handful of values — "
|
|
775
|
-
"consider dropping it or revisiting upstream featurization.",
|
|
776
|
-
series.name,
|
|
777
|
-
mode_fraction,
|
|
778
|
-
n_distinct,
|
|
779
|
-
)
|
|
780
797
|
return entry
|
|
781
798
|
|
|
782
799
|
# Count with a non-zero mode. Linear+clip on each side of the mode,
|
|
@@ -889,9 +906,7 @@ def _apply_constant(
|
|
|
889
906
|
return _to_output_array(out, output_dtype)
|
|
890
907
|
|
|
891
908
|
|
|
892
|
-
def _apply_binary(
|
|
893
|
-
series: pd.Series, transform: dict, output_dtype: str
|
|
894
|
-
) -> np.ndarray:
|
|
909
|
+
def _apply_binary(series: pd.Series, transform: dict, output_dtype: str) -> np.ndarray:
|
|
895
910
|
low = float(transform["low"])
|
|
896
911
|
high = float(transform["high"])
|
|
897
912
|
arr = series.to_numpy(dtype=float)
|
|
@@ -1123,9 +1138,7 @@ def _apply_continuous_left(
|
|
|
1123
1138
|
else: # "finite"
|
|
1124
1139
|
low = float(transform["low_anchor"])
|
|
1125
1140
|
max_extent = high - low
|
|
1126
|
-
out[tail_mask] = -_quadratic_tail(
|
|
1127
|
-
mag, body_span, body_target, max_extent
|
|
1128
|
-
)
|
|
1141
|
+
out[tail_mask] = -_quadratic_tail(mag, body_span, body_target, max_extent)
|
|
1129
1142
|
|
|
1130
1143
|
# Cubic Hermite blend across the body→middle junction (mirror of
|
|
1131
1144
|
# the right-skew blend).
|
|
@@ -1193,7 +1206,9 @@ def _apply_continuous_centered(
|
|
|
1193
1206
|
nan = np.isnan(arr)
|
|
1194
1207
|
out = np.full(arr.shape, np.nan, dtype=float)
|
|
1195
1208
|
|
|
1196
|
-
def _side(
|
|
1209
|
+
def _side(
|
|
1210
|
+
mask: np.ndarray, body_extent: float, max_extent: float, sign: float
|
|
1211
|
+
) -> None:
|
|
1197
1212
|
if not mask.any():
|
|
1198
1213
|
return
|
|
1199
1214
|
if body_extent <= 0:
|
|
@@ -1249,6 +1264,73 @@ def _dispatch_apply(
|
|
|
1249
1264
|
# ---------------------------------------------------------------------------
|
|
1250
1265
|
|
|
1251
1266
|
|
|
1267
|
+
def _fit_series(series: pd.Series) -> dict:
|
|
1268
|
+
"""Classify one column and fit its type-specific transform.
|
|
1269
|
+
|
|
1270
|
+
Shared by the in-memory :func:`fit` and the streaming
|
|
1271
|
+
:func:`fit_file`. All-NaN columns and columns that slip past
|
|
1272
|
+
classification without a usable scale both fall back to
|
|
1273
|
+
``kind: "constant"``.
|
|
1274
|
+
"""
|
|
1275
|
+
if series.dropna().empty:
|
|
1276
|
+
# All-NaN column: fit as a constant. The dispatch at transform
|
|
1277
|
+
# time maps non-NaN inputs to 0 and propagates NaN.
|
|
1278
|
+
return _fit_constant(series)
|
|
1279
|
+
type_ = _classify_type(series)
|
|
1280
|
+
if type_ == "constant":
|
|
1281
|
+
return _fit_constant(series)
|
|
1282
|
+
if type_ == "binary":
|
|
1283
|
+
return _fit_binary(series)
|
|
1284
|
+
if type_ == "count":
|
|
1285
|
+
return _fit_count(series)
|
|
1286
|
+
try:
|
|
1287
|
+
return _fit_continuous(series)
|
|
1288
|
+
except EosframesError:
|
|
1289
|
+
# Column slipped past _classify_type but has no usable scale.
|
|
1290
|
+
return _fit_constant(series)
|
|
1291
|
+
|
|
1292
|
+
|
|
1293
|
+
def _log_fit_summary(columns: dict, logger) -> None:
|
|
1294
|
+
"""Emit the per-fit kind breakdown and near-degenerate advisory.
|
|
1295
|
+
|
|
1296
|
+
Shared by :func:`fit` and :func:`fit_file` so the streaming and
|
|
1297
|
+
in-memory paths log identically.
|
|
1298
|
+
"""
|
|
1299
|
+
kind_counts: dict = {}
|
|
1300
|
+
for entry in columns.values():
|
|
1301
|
+
kind = entry["transform"]["kind"]
|
|
1302
|
+
kind_counts[kind] = kind_counts.get(kind, 0) + 1
|
|
1303
|
+
|
|
1304
|
+
kind_breakdown = ", ".join(
|
|
1305
|
+
f"{kind}={count}" for kind, count in sorted(kind_counts.items())
|
|
1306
|
+
)
|
|
1307
|
+
n = len(columns)
|
|
1308
|
+
logger.info("Fitted %d / %d numeric columns (%s)", n, n, kind_breakdown)
|
|
1309
|
+
|
|
1310
|
+
# Single aggregated advisory for near-degenerate count columns. The
|
|
1311
|
+
# per-column flag lives in each entry's fit_notes; here we just
|
|
1312
|
+
# summarise how many tripped it so sparse fingerprints don't flood
|
|
1313
|
+
# the log with one warning per bit.
|
|
1314
|
+
degenerate_cols = [
|
|
1315
|
+
col
|
|
1316
|
+
for col, entry in columns.items()
|
|
1317
|
+
if entry.get("fit_notes", {}).get("degenerate")
|
|
1318
|
+
]
|
|
1319
|
+
if degenerate_cols:
|
|
1320
|
+
preview = ", ".join(str(c) for c in degenerate_cols[:5])
|
|
1321
|
+
if len(degenerate_cols) > 5:
|
|
1322
|
+
preview += ", …"
|
|
1323
|
+
logger.warning(
|
|
1324
|
+
"%d of %d count columns are near-degenerate (e.g. %s): output "
|
|
1325
|
+
"collapses to a handful of values. Common for sparse "
|
|
1326
|
+
"fingerprints; see each column's fit_notes for details. "
|
|
1327
|
+
"Consider dropping them or revisiting upstream featurization.",
|
|
1328
|
+
len(degenerate_cols),
|
|
1329
|
+
n,
|
|
1330
|
+
preview,
|
|
1331
|
+
)
|
|
1332
|
+
|
|
1333
|
+
|
|
1252
1334
|
def fit(df: pd.DataFrame) -> dict:
|
|
1253
1335
|
"""Fit a type-aware robust scaler on the numeric feature columns.
|
|
1254
1336
|
|
|
@@ -1294,44 +1376,10 @@ def fit(df: pd.DataFrame) -> dict:
|
|
|
1294
1376
|
raise EosframesError("No numeric feature columns found to fit the scaler.")
|
|
1295
1377
|
|
|
1296
1378
|
columns: dict = {}
|
|
1297
|
-
|
|
1298
|
-
|
|
1299
|
-
for col in numeric_cols:
|
|
1300
|
-
series = df[col]
|
|
1301
|
-
|
|
1302
|
-
if series.dropna().empty:
|
|
1303
|
-
# All-NaN column: fit as a constant. The dispatch at
|
|
1304
|
-
# transform time maps non-NaN inputs to 0 and propagates NaN.
|
|
1305
|
-
entry = _fit_constant(series)
|
|
1306
|
-
else:
|
|
1307
|
-
type_ = _classify_type(series)
|
|
1308
|
-
if type_ == "constant":
|
|
1309
|
-
entry = _fit_constant(series)
|
|
1310
|
-
elif type_ == "binary":
|
|
1311
|
-
entry = _fit_binary(series)
|
|
1312
|
-
elif type_ == "count":
|
|
1313
|
-
entry = _fit_count(series)
|
|
1314
|
-
else:
|
|
1315
|
-
try:
|
|
1316
|
-
entry = _fit_continuous(series)
|
|
1317
|
-
except EosframesError:
|
|
1318
|
-
# Column slipped past _classify_type but has no usable scale.
|
|
1319
|
-
entry = _fit_constant(series)
|
|
1320
|
-
|
|
1321
|
-
columns[col] = entry
|
|
1322
|
-
kind = entry["transform"]["kind"]
|
|
1323
|
-
kind_counts[kind] = kind_counts.get(kind, 0) + 1
|
|
1324
|
-
|
|
1325
|
-
kind_breakdown = ", ".join(
|
|
1326
|
-
f"{kind}={count}" for kind, count in sorted(kind_counts.items())
|
|
1327
|
-
)
|
|
1328
|
-
logger.info(
|
|
1329
|
-
"Fitted %d / %d numeric columns (%s)",
|
|
1330
|
-
len(numeric_cols),
|
|
1331
|
-
len(numeric_cols),
|
|
1332
|
-
kind_breakdown,
|
|
1333
|
-
)
|
|
1379
|
+
for col in progress(numeric_cols, desc="Fitting columns"):
|
|
1380
|
+
columns[col] = _fit_series(df[col])
|
|
1334
1381
|
|
|
1382
|
+
_log_fit_summary(columns, logger)
|
|
1335
1383
|
return {"method": _METHOD_NAME, "columns": columns}
|
|
1336
1384
|
|
|
1337
1385
|
|
|
@@ -1381,10 +1429,15 @@ def transform(
|
|
|
1381
1429
|
On column mismatch, invalid ``output_dtype``, or unknown
|
|
1382
1430
|
column type in *params*.
|
|
1383
1431
|
"""
|
|
1432
|
+
_validate_transform(df, params, output_dtype)
|
|
1433
|
+
return _apply_transform(df, params, output_dtype, impute, show_progress=True)
|
|
1434
|
+
|
|
1435
|
+
|
|
1436
|
+
def _validate_transform(df: pd.DataFrame, params: dict, output_dtype: str) -> None:
|
|
1437
|
+
"""Check output dtype and exact feature-column match before applying."""
|
|
1384
1438
|
if output_dtype not in _VALID_OUTPUT_DTYPES:
|
|
1385
1439
|
raise EosframesError(
|
|
1386
|
-
f"Unknown output_dtype '{output_dtype}'. "
|
|
1387
|
-
f"Supported: {_VALID_OUTPUT_DTYPES}"
|
|
1440
|
+
f"Unknown output_dtype '{output_dtype}'. Supported: {_VALID_OUTPUT_DTYPES}"
|
|
1388
1441
|
)
|
|
1389
1442
|
|
|
1390
1443
|
expected_feature_cols = list(params["columns"].keys())
|
|
@@ -1396,9 +1449,29 @@ def transform(
|
|
|
1396
1449
|
f"but transformer was fitted on {expected_feature_cols}."
|
|
1397
1450
|
)
|
|
1398
1451
|
|
|
1452
|
+
|
|
1453
|
+
def _apply_transform(
|
|
1454
|
+
df: pd.DataFrame,
|
|
1455
|
+
params: dict,
|
|
1456
|
+
output_dtype: str,
|
|
1457
|
+
impute: bool,
|
|
1458
|
+
show_progress: bool,
|
|
1459
|
+
) -> pd.DataFrame:
|
|
1460
|
+
"""Apply the fitted per-column transforms to *df* (already validated).
|
|
1461
|
+
|
|
1462
|
+
*show_progress* draws the per-column bar — on for the single-shot
|
|
1463
|
+
in-memory :func:`transform`, off for the streaming path where the bar
|
|
1464
|
+
would otherwise redraw once per row-chunk.
|
|
1465
|
+
"""
|
|
1399
1466
|
columns = params["columns"]
|
|
1467
|
+
expected_feature_cols = list(columns.keys())
|
|
1400
1468
|
result = df.copy()
|
|
1401
|
-
|
|
1469
|
+
iterator = (
|
|
1470
|
+
progress(expected_feature_cols, desc="Transforming columns")
|
|
1471
|
+
if show_progress
|
|
1472
|
+
else expected_feature_cols
|
|
1473
|
+
)
|
|
1474
|
+
for col in iterator:
|
|
1402
1475
|
entry = columns[col]
|
|
1403
1476
|
series = df[col]
|
|
1404
1477
|
if impute:
|
|
@@ -1418,32 +1491,260 @@ def _values_dtype_for(output_dtype: str) -> np.dtype:
|
|
|
1418
1491
|
return np.float32
|
|
1419
1492
|
|
|
1420
1493
|
|
|
1421
|
-
|
|
1422
|
-
|
|
1423
|
-
|
|
1424
|
-
|
|
1494
|
+
# ---------------------------------------------------------------------------
|
|
1495
|
+
# Streaming primitives — bounded-memory fit/transform for very wide or very
|
|
1496
|
+
# tall files. The whole matrix is never resident: fit walks one column at a
|
|
1497
|
+
# time (cheap on the columnar H5 layout), transform walks row-chunks.
|
|
1498
|
+
# ---------------------------------------------------------------------------
|
|
1499
|
+
|
|
1500
|
+
|
|
1501
|
+
def _h5_feature_names(h5_path: str) -> list:
|
|
1502
|
+
with h5py.File(h5_path, "r") as f:
|
|
1503
|
+
return [x.decode("utf-8") for x in f["features"][:]]
|
|
1504
|
+
|
|
1505
|
+
|
|
1506
|
+
def _h5_nrows(h5_path: str) -> int:
|
|
1507
|
+
with h5py.File(h5_path, "r") as f:
|
|
1508
|
+
return int(f["values"].shape[0])
|
|
1509
|
+
|
|
1510
|
+
|
|
1511
|
+
def _iter_h5_columns(
|
|
1512
|
+
h5_path: str, batch: int = _FIT_COLUMN_BATCH
|
|
1513
|
+
) -> Iterator[Tuple[str, np.ndarray]]:
|
|
1514
|
+
"""Yield ``(feature_name, full_column_array)`` for every feature column.
|
|
1515
|
+
|
|
1516
|
+
Columns are pulled from disk a *batch* at a time — one ``values[:,
|
|
1517
|
+
start:stop]`` hyperslab — then handed out one at a time from that
|
|
1518
|
+
in-memory block. This keeps peak memory bounded (``batch`` columns, not
|
|
1519
|
+
the whole matrix) while reading each underlying HDF5 storage chunk once,
|
|
1520
|
+
instead of re-reading it for every column it contains.
|
|
1521
|
+
"""
|
|
1522
|
+
with h5py.File(h5_path, "r") as f:
|
|
1523
|
+
features = [x.decode("utf-8") for x in f["features"][:]]
|
|
1524
|
+
values = f["values"]
|
|
1525
|
+
n_features = len(features)
|
|
1526
|
+
for start in range(0, n_features, batch):
|
|
1527
|
+
block = values[:, start : start + batch] # (n_rows, ≤batch)
|
|
1528
|
+
for k in range(block.shape[1]):
|
|
1529
|
+
yield features[start + k], block[:, k]
|
|
1530
|
+
|
|
1531
|
+
|
|
1532
|
+
def _stage_csv_to_h5(csv_path: str, h5_path: str, chunksize: int) -> Tuple[list, int]:
|
|
1533
|
+
"""Stream a CSV into a columnar temp H5 in one bounded-memory pass.
|
|
1534
|
+
|
|
1535
|
+
Reads the CSV in row-chunks and appends each chunk's numeric feature
|
|
1536
|
+
columns to a resizable ``values`` dataset (``float32`` — matches the
|
|
1537
|
+
pipeline's default output precision). Returns ``(feature_cols, n_rows)``.
|
|
1538
|
+
The resulting H5 is then cheaply column-sliceable for the fit.
|
|
1539
|
+
"""
|
|
1540
|
+
logger = get_logger()
|
|
1541
|
+
feature_cols: Optional[list] = None
|
|
1542
|
+
total = 0
|
|
1543
|
+
next_log = _STAGE_LOG_ROWS
|
|
1544
|
+
with h5py.File(h5_path, "w") as f:
|
|
1545
|
+
values_ds = None
|
|
1546
|
+
for chunk in pd.read_csv(csv_path, chunksize=chunksize):
|
|
1547
|
+
if feature_cols is None:
|
|
1548
|
+
feat = [c for c in chunk.columns if c not in _META_COLS]
|
|
1549
|
+
feature_cols = [
|
|
1550
|
+
c for c in feat if pd.api.types.is_numeric_dtype(chunk[c])
|
|
1551
|
+
]
|
|
1552
|
+
n_feat = len(feature_cols)
|
|
1553
|
+
# Tile chunks to the fit's column-batch width so each storage
|
|
1554
|
+
# chunk is read exactly once during the (batched) column fit.
|
|
1555
|
+
if n_feat > 0:
|
|
1556
|
+
col_tile = min(n_feat, _FIT_COLUMN_BATCH)
|
|
1557
|
+
row_tile = max(1, _STAGE_CHUNK_BYTES // (col_tile * 4))
|
|
1558
|
+
chunk_shape: object = (row_tile, col_tile)
|
|
1559
|
+
else:
|
|
1560
|
+
chunk_shape = True
|
|
1561
|
+
values_ds = f.create_dataset(
|
|
1562
|
+
"values",
|
|
1563
|
+
shape=(0, n_feat),
|
|
1564
|
+
maxshape=(None, n_feat),
|
|
1565
|
+
dtype=np.float32,
|
|
1566
|
+
chunks=chunk_shape,
|
|
1567
|
+
)
|
|
1568
|
+
logger.info("Staging: detected %d numeric feature columns", n_feat)
|
|
1569
|
+
arr = chunk[feature_cols].to_numpy(dtype=np.float32)
|
|
1570
|
+
n = arr.shape[0]
|
|
1571
|
+
values_ds.resize(total + n, axis=0)
|
|
1572
|
+
values_ds[total : total + n, :] = arr
|
|
1573
|
+
total += n
|
|
1574
|
+
if total >= next_log:
|
|
1575
|
+
logger.info("Staging: %s rows written to temp H5", f"{total:,}")
|
|
1576
|
+
next_log += _STAGE_LOG_ROWS
|
|
1577
|
+
dt = h5py.string_dtype(encoding="utf-8")
|
|
1578
|
+
f.create_dataset("features", data=feature_cols or [], dtype=dt)
|
|
1579
|
+
logger.info(
|
|
1580
|
+
"Staging complete: %s rows × %d columns → %s",
|
|
1581
|
+
f"{total:,}",
|
|
1582
|
+
len(feature_cols or []),
|
|
1583
|
+
os.path.basename(h5_path),
|
|
1584
|
+
)
|
|
1585
|
+
return feature_cols or [], total
|
|
1586
|
+
|
|
1587
|
+
|
|
1588
|
+
def _iter_row_chunks(path: str, chunksize: int) -> Iterator[pd.DataFrame]:
|
|
1589
|
+
"""Yield row-chunk DataFrames (``key``/``input`` + feature cols) from a file.
|
|
1425
1590
|
|
|
1426
|
-
|
|
1591
|
+
CSV is streamed natively by pandas; H5 is sliced ``[start:end, :]`` so
|
|
1592
|
+
only ``chunksize`` rows are resident at a time. The yielded frames have
|
|
1593
|
+
the same column layout the in-memory :func:`transform` expects.
|
|
1427
1594
|
"""
|
|
1428
|
-
ext = os.path.splitext(
|
|
1595
|
+
ext = os.path.splitext(path)[1].lower()
|
|
1429
1596
|
if ext == ".csv":
|
|
1430
|
-
|
|
1597
|
+
yield from pd.read_csv(path, chunksize=chunksize)
|
|
1431
1598
|
elif ext == ".h5":
|
|
1432
|
-
|
|
1433
|
-
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1441
|
-
|
|
1442
|
-
|
|
1443
|
-
|
|
1444
|
-
|
|
1599
|
+
with h5py.File(path, "r") as f:
|
|
1600
|
+
features = [x.decode("utf-8") for x in f["features"][:]]
|
|
1601
|
+
values = f["values"]
|
|
1602
|
+
n_rows = values.shape[0]
|
|
1603
|
+
has_key = "key" in f
|
|
1604
|
+
for start in range(0, n_rows, chunksize):
|
|
1605
|
+
end = min(start + chunksize, n_rows)
|
|
1606
|
+
meta = {}
|
|
1607
|
+
if has_key:
|
|
1608
|
+
meta["key"] = [x.decode("utf-8") for x in f["key"][start:end]]
|
|
1609
|
+
meta["input"] = [x.decode("utf-8") for x in f["input"][start:end]]
|
|
1610
|
+
block = pd.DataFrame(values[start:end, :], columns=features)
|
|
1611
|
+
yield pd.concat([pd.DataFrame(meta), block], axis=1)
|
|
1445
1612
|
else:
|
|
1446
|
-
raise EosframesError(f"Unsupported
|
|
1613
|
+
raise EosframesError(f"Unsupported input format '{ext}'. Expected .csv or .h5")
|
|
1614
|
+
|
|
1615
|
+
|
|
1616
|
+
class _StreamWriter:
|
|
1617
|
+
"""Incremental CSV/H5 writer — appends row-chunks without holding the file.
|
|
1618
|
+
|
|
1619
|
+
CSV: header on the first chunk, append thereafter. H5: resizable
|
|
1620
|
+
``values`` / ``input`` / ``key`` datasets created from the first chunk's
|
|
1621
|
+
schema, then extended per chunk. Bypasses the naming-convention check
|
|
1622
|
+
(the caller has already validated the output path).
|
|
1623
|
+
"""
|
|
1624
|
+
|
|
1625
|
+
def __init__(self, path: str, values_dtype: np.dtype):
|
|
1626
|
+
self.path = path
|
|
1627
|
+
self.ext = os.path.splitext(path)[1].lower()
|
|
1628
|
+
self.values_dtype = values_dtype
|
|
1629
|
+
if self.ext not in (".csv", ".h5"):
|
|
1630
|
+
raise EosframesError(
|
|
1631
|
+
f"Unsupported output format '{self.ext}'. Expected .csv or .h5"
|
|
1632
|
+
)
|
|
1633
|
+
self._csv_started = False
|
|
1634
|
+
self._h5 = None
|
|
1635
|
+
self._feature_cols: Optional[list] = None
|
|
1636
|
+
self._has_key = False
|
|
1637
|
+
self._n = 0
|
|
1638
|
+
|
|
1639
|
+
def write(self, df: pd.DataFrame) -> None:
|
|
1640
|
+
if self.ext == ".csv":
|
|
1641
|
+
df.to_csv(
|
|
1642
|
+
self.path,
|
|
1643
|
+
mode="w" if not self._csv_started else "a",
|
|
1644
|
+
header=not self._csv_started,
|
|
1645
|
+
index=False,
|
|
1646
|
+
)
|
|
1647
|
+
self._csv_started = True
|
|
1648
|
+
else:
|
|
1649
|
+
if self._h5 is None:
|
|
1650
|
+
self._init_h5(df)
|
|
1651
|
+
self._append_h5(df)
|
|
1652
|
+
|
|
1653
|
+
def _init_h5(self, df: pd.DataFrame) -> None:
|
|
1654
|
+
self._feature_cols = [c for c in df.columns if c not in _META_COLS]
|
|
1655
|
+
self._has_key = "key" in df.columns
|
|
1656
|
+
self._h5 = h5py.File(self.path, "w")
|
|
1657
|
+
dt = h5py.string_dtype(encoding="utf-8")
|
|
1658
|
+
self._h5.create_dataset("features", data=self._feature_cols, dtype=dt)
|
|
1659
|
+
self._values = self._h5.create_dataset(
|
|
1660
|
+
"values",
|
|
1661
|
+
shape=(0, len(self._feature_cols)),
|
|
1662
|
+
maxshape=(None, len(self._feature_cols)),
|
|
1663
|
+
dtype=self.values_dtype,
|
|
1664
|
+
chunks=True,
|
|
1665
|
+
)
|
|
1666
|
+
self._input = self._h5.create_dataset(
|
|
1667
|
+
"input", shape=(0,), maxshape=(None,), dtype=dt
|
|
1668
|
+
)
|
|
1669
|
+
if self._has_key:
|
|
1670
|
+
self._key = self._h5.create_dataset(
|
|
1671
|
+
"key", shape=(0,), maxshape=(None,), dtype=dt
|
|
1672
|
+
)
|
|
1673
|
+
|
|
1674
|
+
def _append_h5(self, df: pd.DataFrame) -> None:
|
|
1675
|
+
n = len(df)
|
|
1676
|
+
new = self._n + n
|
|
1677
|
+
self._values.resize(new, axis=0)
|
|
1678
|
+
self._values[self._n : new, :] = df[self._feature_cols].to_numpy(
|
|
1679
|
+
dtype=self.values_dtype
|
|
1680
|
+
)
|
|
1681
|
+
self._input.resize(new, axis=0)
|
|
1682
|
+
self._input[self._n : new] = df["input"].astype(str).tolist()
|
|
1683
|
+
if self._has_key:
|
|
1684
|
+
self._key.resize(new, axis=0)
|
|
1685
|
+
self._key[self._n : new] = df["key"].astype(str).tolist()
|
|
1686
|
+
self._n = new
|
|
1687
|
+
|
|
1688
|
+
def close(self) -> None:
|
|
1689
|
+
if self._h5 is not None:
|
|
1690
|
+
self._h5.close()
|
|
1691
|
+
self._h5 = None
|
|
1692
|
+
|
|
1693
|
+
|
|
1694
|
+
def _streaming_transform(
|
|
1695
|
+
input_path: str,
|
|
1696
|
+
transformer: dict,
|
|
1697
|
+
output_path: str,
|
|
1698
|
+
output_dtype: str,
|
|
1699
|
+
impute: bool,
|
|
1700
|
+
chunksize: int,
|
|
1701
|
+
) -> int:
|
|
1702
|
+
"""Row-chunked transform: read a chunk, scale it, write it, repeat.
|
|
1703
|
+
|
|
1704
|
+
Reuses the in-memory per-column dispatch on each small chunk (with its
|
|
1705
|
+
per-column bar suppressed), so peak memory is one chunk in + one chunk
|
|
1706
|
+
out — independent of file size. Returns the number of rows written.
|
|
1707
|
+
"""
|
|
1708
|
+
logger = get_logger()
|
|
1709
|
+
in_ext = os.path.splitext(input_path)[1].lower()
|
|
1710
|
+
# Row count is free for H5 (so we get a true % bar / steady log cadence);
|
|
1711
|
+
# for CSV it is unknown without a full scan, so we log every fixed number
|
|
1712
|
+
# of chunks instead.
|
|
1713
|
+
total_chunks = None
|
|
1714
|
+
if in_ext == ".h5":
|
|
1715
|
+
total_chunks = (_h5_nrows(input_path) + chunksize - 1) // chunksize
|
|
1716
|
+
logger.info(
|
|
1717
|
+
"Transforming %s rows in %d chunks of %d",
|
|
1718
|
+
f"{_h5_nrows(input_path):,}",
|
|
1719
|
+
total_chunks,
|
|
1720
|
+
chunksize,
|
|
1721
|
+
)
|
|
1722
|
+
log_every = (
|
|
1723
|
+
max(1, total_chunks // _PROGRESS_LOG_STEPS)
|
|
1724
|
+
if total_chunks
|
|
1725
|
+
else _PROGRESS_LOG_STEPS
|
|
1726
|
+
)
|
|
1727
|
+
|
|
1728
|
+
writer = _StreamWriter(output_path, _values_dtype_for(output_dtype))
|
|
1729
|
+
chunks = progress(
|
|
1730
|
+
_iter_row_chunks(input_path, chunksize),
|
|
1731
|
+
total=total_chunks,
|
|
1732
|
+
desc="Transforming chunks",
|
|
1733
|
+
log_every=log_every,
|
|
1734
|
+
)
|
|
1735
|
+
|
|
1736
|
+
total_rows = 0
|
|
1737
|
+
try:
|
|
1738
|
+
for chunk in chunks:
|
|
1739
|
+
_validate_transform(chunk, transformer, output_dtype)
|
|
1740
|
+
scaled = _apply_transform(
|
|
1741
|
+
chunk, transformer, output_dtype, impute, show_progress=False
|
|
1742
|
+
)
|
|
1743
|
+
writer.write(scaled)
|
|
1744
|
+
total_rows += len(scaled)
|
|
1745
|
+
finally:
|
|
1746
|
+
writer.close()
|
|
1747
|
+
return total_rows
|
|
1447
1748
|
|
|
1448
1749
|
|
|
1449
1750
|
def fit_file(
|
|
@@ -1452,15 +1753,27 @@ def fit_file(
|
|
|
1452
1753
|
output_path: Optional[str] = None,
|
|
1453
1754
|
output_dtype: str = _DEFAULT_OUTPUT_DTYPE,
|
|
1454
1755
|
impute: bool = False,
|
|
1756
|
+
chunksize: int = _DEFAULT_CHUNKSIZE,
|
|
1455
1757
|
) -> str:
|
|
1456
1758
|
"""Fit a scaler on an Ersilia output file and save the parameters.
|
|
1457
1759
|
|
|
1760
|
+
**Bounded memory.** The full matrix is never resident. The fit reads a
|
|
1761
|
+
batch of feature columns at a time (``_FIT_COLUMN_BATCH``) and fits each
|
|
1762
|
+
column from that in-memory block, so peak memory is the batch
|
|
1763
|
+
(~``batch × n_rows`` floats) — a few hundred MB — regardless of how wide
|
|
1764
|
+
the file is. H5 inputs are sliced directly; CSV inputs are first
|
|
1765
|
+
streamed — in
|
|
1766
|
+
row-chunks of *chunksize* — into a temporary columnar H5 (``float32``),
|
|
1767
|
+
which is then column-sliced for the fit and removed afterwards; this is
|
|
1768
|
+
the one unavoidable full read of the CSV, done without loading it whole.
|
|
1769
|
+
|
|
1458
1770
|
The scaler JSON written to *scaler_path* is dtype-agnostic — see
|
|
1459
1771
|
:func:`fit` for the parameter set. When *output_path* is provided
|
|
1460
1772
|
the scaled data is also written immediately (fit-then-transform in
|
|
1461
|
-
one call), and *output_dtype* selects
|
|
1462
|
-
output. The dtype is **not** recorded in the
|
|
1463
|
-
calls to :func:`transform_file` choose the dtype
|
|
1773
|
+
one call, streamed row-chunk by row-chunk), and *output_dtype* selects
|
|
1774
|
+
the dtype of that inline output. The dtype is **not** recorded in the
|
|
1775
|
+
scaler JSON; later calls to :func:`transform_file` choose the dtype
|
|
1776
|
+
independently.
|
|
1464
1777
|
|
|
1465
1778
|
The scaler filename's encoded model ID and version must match the
|
|
1466
1779
|
input file's. The transformer JSON records ``eosframes_version``
|
|
@@ -1487,6 +1800,10 @@ def fit_file(
|
|
|
1487
1800
|
Only used for the inline transform when *output_path* is
|
|
1488
1801
|
given. ``"int8"`` quantizes the scaled values into
|
|
1489
1802
|
``[-127, 127]`` with sentinel ``-128`` for missing.
|
|
1803
|
+
chunksize : int, default ``50_000``
|
|
1804
|
+
Rows per chunk for the CSV→H5 staging pass and the inline
|
|
1805
|
+
transform. Bounds peak memory; tune down for extremely wide
|
|
1806
|
+
frames, up for narrow ones.
|
|
1490
1807
|
|
|
1491
1808
|
Returns
|
|
1492
1809
|
-------
|
|
@@ -1540,30 +1857,71 @@ def fit_file(
|
|
|
1540
1857
|
f"Output file '{output_path}' already exists. Remove it first."
|
|
1541
1858
|
)
|
|
1542
1859
|
|
|
1543
|
-
from .ops import _read_file
|
|
1544
|
-
|
|
1545
|
-
df = _read_file(input_path)
|
|
1546
|
-
|
|
1547
1860
|
if output_dtype not in _VALID_OUTPUT_DTYPES:
|
|
1548
1861
|
raise EosframesError(
|
|
1549
|
-
f"Unknown output_dtype '{output_dtype}'. "
|
|
1550
|
-
f"Supported: {_VALID_OUTPUT_DTYPES}"
|
|
1862
|
+
f"Unknown output_dtype '{output_dtype}'. Supported: {_VALID_OUTPUT_DTYPES}"
|
|
1551
1863
|
)
|
|
1552
1864
|
|
|
1865
|
+
in_ext = os.path.splitext(input_path)[1].lower()
|
|
1553
1866
|
mode_label = "fit + transform" if output_path is not None else "fit only"
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1867
|
+
|
|
1868
|
+
# Resolve a column-sliceable H5 to fit from. H5 inputs are used as-is;
|
|
1869
|
+
# CSV inputs are streamed into a temporary columnar H5 first.
|
|
1870
|
+
staged_tmp: Optional[str] = None
|
|
1871
|
+
try:
|
|
1872
|
+
if in_ext == ".h5":
|
|
1873
|
+
fit_source = input_path
|
|
1874
|
+
feature_cols = _h5_feature_names(input_path)
|
|
1875
|
+
n_rows = _h5_nrows(input_path)
|
|
1876
|
+
elif in_ext == ".csv":
|
|
1877
|
+
fd, staged_tmp = tempfile.mkstemp(
|
|
1878
|
+
suffix=".h5", dir=os.path.dirname(os.path.abspath(input_path))
|
|
1879
|
+
)
|
|
1880
|
+
os.close(fd)
|
|
1881
|
+
logger.info(
|
|
1882
|
+
"%s — staging CSV → temporary columnar H5 for column-wise fit "
|
|
1883
|
+
"(chunksize=%d)",
|
|
1884
|
+
mode_label,
|
|
1885
|
+
chunksize,
|
|
1886
|
+
)
|
|
1887
|
+
feature_cols, n_rows = _stage_csv_to_h5(input_path, staged_tmp, chunksize)
|
|
1888
|
+
fit_source = staged_tmp
|
|
1889
|
+
else:
|
|
1890
|
+
raise EosframesError(
|
|
1891
|
+
f"Unsupported input format '{in_ext}'. Expected .csv or .h5"
|
|
1892
|
+
)
|
|
1893
|
+
|
|
1894
|
+
if not feature_cols:
|
|
1895
|
+
raise EosframesError("No numeric feature columns found to fit the scaler.")
|
|
1896
|
+
|
|
1897
|
+
logger.info(
|
|
1898
|
+
"%s — fitting %d columns over %s rows (batches of %d)",
|
|
1899
|
+
mode_label,
|
|
1900
|
+
len(feature_cols),
|
|
1901
|
+
f"{n_rows:,}",
|
|
1902
|
+
_FIT_COLUMN_BATCH,
|
|
1903
|
+
)
|
|
1904
|
+
columns: dict = {}
|
|
1905
|
+
for name, arr in progress(
|
|
1906
|
+
_iter_h5_columns(fit_source),
|
|
1907
|
+
total=len(feature_cols),
|
|
1908
|
+
desc="Fitting columns",
|
|
1909
|
+
log_every=max(1, len(feature_cols) // _PROGRESS_LOG_STEPS),
|
|
1910
|
+
):
|
|
1911
|
+
columns[name] = _fit_series(pd.Series(arr))
|
|
1912
|
+
_log_fit_summary(columns, logger)
|
|
1913
|
+
finally:
|
|
1914
|
+
if staged_tmp is not None and os.path.exists(staged_tmp):
|
|
1915
|
+
os.remove(staged_tmp)
|
|
1558
1916
|
|
|
1559
1917
|
transformer = {
|
|
1560
1918
|
"eosframes_version": _PACKAGE_VERSION,
|
|
1561
|
-
"method":
|
|
1919
|
+
"method": _METHOD_NAME,
|
|
1562
1920
|
"model_id": parsed["model_id"],
|
|
1563
1921
|
"model_version": parsed["version"],
|
|
1564
1922
|
"fitted_at": datetime.now().isoformat(timespec="seconds"),
|
|
1565
|
-
"n_rows":
|
|
1566
|
-
"columns":
|
|
1923
|
+
"n_rows": n_rows,
|
|
1924
|
+
"columns": columns,
|
|
1567
1925
|
}
|
|
1568
1926
|
with open(scaler_path, "w") as fh:
|
|
1569
1927
|
json.dump(transformer, fh, indent=2)
|
|
@@ -1575,15 +1933,15 @@ def fit_file(
|
|
|
1575
1933
|
output_dtype,
|
|
1576
1934
|
impute,
|
|
1577
1935
|
)
|
|
1578
|
-
|
|
1579
|
-
|
|
1580
|
-
|
|
1581
|
-
_write_df(
|
|
1582
|
-
scaled_df,
|
|
1936
|
+
written = _streaming_transform(
|
|
1937
|
+
input_path,
|
|
1938
|
+
transformer,
|
|
1583
1939
|
output_path,
|
|
1584
|
-
|
|
1940
|
+
output_dtype,
|
|
1941
|
+
impute,
|
|
1942
|
+
chunksize,
|
|
1585
1943
|
)
|
|
1586
|
-
logger.info("Scaled output written to %s", output_path)
|
|
1944
|
+
logger.info("Scaled output written to %s (%d rows)", output_path, written)
|
|
1587
1945
|
|
|
1588
1946
|
return scaler_path
|
|
1589
1947
|
|
|
@@ -1594,9 +1952,15 @@ def transform_file(
|
|
|
1594
1952
|
output_path: str,
|
|
1595
1953
|
output_dtype: str = _DEFAULT_OUTPUT_DTYPE,
|
|
1596
1954
|
impute: bool = False,
|
|
1955
|
+
chunksize: int = _DEFAULT_CHUNKSIZE,
|
|
1597
1956
|
) -> str:
|
|
1598
1957
|
"""Apply a saved scaler to an Ersilia output file.
|
|
1599
1958
|
|
|
1959
|
+
**Bounded memory.** The input is read and the output written one
|
|
1960
|
+
row-chunk of *chunksize* at a time, so peak memory is a single chunk
|
|
1961
|
+
in plus a single chunk out — independent of the file's size. Works for
|
|
1962
|
+
CSV and H5 inputs and outputs in any combination.
|
|
1963
|
+
|
|
1600
1964
|
The scaler's recorded ``eosframes_version`` must exactly match the
|
|
1601
1965
|
running ``eosframes.__version__``, and ``method`` must still be
|
|
1602
1966
|
``"robust_typed"``; any mismatch raises with a clear "re-fit"
|
|
@@ -1618,6 +1982,9 @@ def transform_file(
|
|
|
1618
1982
|
output_dtype : {"float32", "int8"}, default ``"float32"``
|
|
1619
1983
|
``"int8"`` quantizes scaled values to ``[-127, 127]`` with
|
|
1620
1984
|
sentinel ``-128`` for missing.
|
|
1985
|
+
chunksize : int, default ``50_000``
|
|
1986
|
+
Rows per streamed chunk. Bounds peak memory; tune down for
|
|
1987
|
+
extremely wide frames, up for narrow ones.
|
|
1621
1988
|
|
|
1622
1989
|
Returns
|
|
1623
1990
|
-------
|
|
@@ -1673,8 +2040,7 @@ def transform_file(
|
|
|
1673
2040
|
|
|
1674
2041
|
if output_dtype not in _VALID_OUTPUT_DTYPES:
|
|
1675
2042
|
raise EosframesError(
|
|
1676
|
-
f"Unknown output_dtype '{output_dtype}'. "
|
|
1677
|
-
f"Supported: {_VALID_OUTPUT_DTYPES}"
|
|
2043
|
+
f"Unknown output_dtype '{output_dtype}'. Supported: {_VALID_OUTPUT_DTYPES}"
|
|
1678
2044
|
)
|
|
1679
2045
|
|
|
1680
2046
|
t_model_id = transformer.get("model_id")
|
|
@@ -1693,21 +2059,21 @@ def transform_file(
|
|
|
1693
2059
|
f"scaler was fitted on '{t_model_version}'."
|
|
1694
2060
|
)
|
|
1695
2061
|
|
|
1696
|
-
from .ops import _read_file
|
|
1697
|
-
|
|
1698
|
-
df = _read_file(input_path)
|
|
1699
|
-
|
|
1700
2062
|
logger.info(
|
|
1701
|
-
"Applying scaler to %
|
|
1702
|
-
|
|
2063
|
+
"Applying scaler to %s in row-chunks (chunksize=%d, output_dtype=%s, "
|
|
2064
|
+
"impute=%s)",
|
|
1703
2065
|
input_path,
|
|
2066
|
+
chunksize,
|
|
1704
2067
|
output_dtype,
|
|
1705
2068
|
impute,
|
|
1706
2069
|
)
|
|
1707
|
-
|
|
1708
|
-
|
|
2070
|
+
written = _streaming_transform(
|
|
2071
|
+
input_path,
|
|
2072
|
+
transformer,
|
|
2073
|
+
output_path,
|
|
2074
|
+
output_dtype,
|
|
2075
|
+
impute,
|
|
2076
|
+
chunksize,
|
|
1709
2077
|
)
|
|
1710
|
-
|
|
1711
|
-
_write_df(scaled_df, output_path, values_dtype=_values_dtype_for(output_dtype))
|
|
1712
|
-
logger.info("Scaled output written to %s", output_path)
|
|
2078
|
+
logger.info("Scaled output written to %s (%d rows)", output_path, written)
|
|
1713
2079
|
return output_path
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""General-purpose utilities."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import sys
|
|
5
|
+
import time
|
|
6
|
+
from typing import Iterable, Iterator, Optional, TypeVar
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from .logger import get_logger
|
|
11
|
+
|
|
12
|
+
_T = TypeVar("_T")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def chunker(df: pd.DataFrame, chunksize: int = 10000) -> Iterator[pd.DataFrame]:
|
|
16
|
+
"""Yield successive non-overlapping chunks of *df*.
|
|
17
|
+
|
|
18
|
+
Parameters
|
|
19
|
+
----------
|
|
20
|
+
df : pd.DataFrame
|
|
21
|
+
The DataFrame to split.
|
|
22
|
+
chunksize : int
|
|
23
|
+
Number of rows per chunk (default 10 000).
|
|
24
|
+
|
|
25
|
+
Yields
|
|
26
|
+
------
|
|
27
|
+
pd.DataFrame
|
|
28
|
+
"""
|
|
29
|
+
for start in range(0, len(df), chunksize):
|
|
30
|
+
yield df.iloc[start : start + chunksize]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def progress(
|
|
34
|
+
iterable: Iterable[_T],
|
|
35
|
+
total: Optional[int] = None,
|
|
36
|
+
desc: str = "",
|
|
37
|
+
log_every: Optional[int] = None,
|
|
38
|
+
) -> Iterator[_T]:
|
|
39
|
+
"""Wrap *iterable* with progress feedback that adapts to the environment.
|
|
40
|
+
|
|
41
|
+
Three modes, picked automatically (all gated on the ``eosframes`` logger
|
|
42
|
+
being at ``INFO`` or more verbose — a quieted logger gets a silent
|
|
43
|
+
pass-through):
|
|
44
|
+
|
|
45
|
+
* **Interactive (TTY)** — an in-place per-item bar on stderr.
|
|
46
|
+
* **Non-interactive with** *log_every* — periodic ``INFO`` log lines
|
|
47
|
+
(every *log_every* items, plus a final line). This is the path that
|
|
48
|
+
keeps long streaming jobs legible when stderr is a file or pipe:
|
|
49
|
+
``nohup``, CI, ``build_scaler.sh``, etc. Without *log_every* a
|
|
50
|
+
non-interactive run stays silent.
|
|
51
|
+
* **Otherwise** — transparent pass-through, no output.
|
|
52
|
+
|
|
53
|
+
Both visible modes share stderr with the logger, so nothing ever leaks
|
|
54
|
+
onto piped stdout.
|
|
55
|
+
|
|
56
|
+
Parameters
|
|
57
|
+
----------
|
|
58
|
+
iterable : Iterable
|
|
59
|
+
The items to iterate over.
|
|
60
|
+
total : int, optional
|
|
61
|
+
Expected number of items. Inferred via ``len(iterable)`` when omitted;
|
|
62
|
+
if neither is available the bar is disabled (percentages are dropped
|
|
63
|
+
from log lines but counts still flow).
|
|
64
|
+
desc : str
|
|
65
|
+
Short label shown to the left of the bar / log line.
|
|
66
|
+
log_every : int, optional
|
|
67
|
+
Emit an ``INFO`` log line every this many items when no bar is drawn.
|
|
68
|
+
Leave unset for the in-memory paths that should stay quiet off-TTY.
|
|
69
|
+
|
|
70
|
+
Yields
|
|
71
|
+
------
|
|
72
|
+
The items of *iterable*, unchanged.
|
|
73
|
+
"""
|
|
74
|
+
if total is None:
|
|
75
|
+
try:
|
|
76
|
+
total = len(iterable) # type: ignore[arg-type]
|
|
77
|
+
except TypeError:
|
|
78
|
+
total = None
|
|
79
|
+
|
|
80
|
+
logger = get_logger()
|
|
81
|
+
info_on = logger.isEnabledFor(logging.INFO)
|
|
82
|
+
bar_on = total is not None and total > 1 and sys.stderr.isatty() and info_on
|
|
83
|
+
|
|
84
|
+
if bar_on:
|
|
85
|
+
width = 30
|
|
86
|
+
last_draw = -1.0 # force an immediate first draw
|
|
87
|
+
count = 0
|
|
88
|
+
for count, item in enumerate(iterable, 1):
|
|
89
|
+
yield item
|
|
90
|
+
now = time.monotonic()
|
|
91
|
+
# Throttle redraws to ~10 fps, but always paint the final item.
|
|
92
|
+
if count < total and now - last_draw < 0.1:
|
|
93
|
+
continue
|
|
94
|
+
last_draw = now
|
|
95
|
+
frac = count / total
|
|
96
|
+
filled = int(width * frac)
|
|
97
|
+
bar = "█" * filled + "·" * (width - filled)
|
|
98
|
+
label = f"{desc} " if desc else ""
|
|
99
|
+
sys.stderr.write(f"\r{label}|{bar}| {count}/{total}")
|
|
100
|
+
sys.stderr.flush()
|
|
101
|
+
if count:
|
|
102
|
+
sys.stderr.write("\n")
|
|
103
|
+
sys.stderr.flush()
|
|
104
|
+
return
|
|
105
|
+
|
|
106
|
+
if log_every and info_on:
|
|
107
|
+
label = desc or "progress"
|
|
108
|
+
count = 0
|
|
109
|
+
for count, item in enumerate(iterable, 1):
|
|
110
|
+
yield item
|
|
111
|
+
if count % log_every == 0:
|
|
112
|
+
if total:
|
|
113
|
+
logger.info(
|
|
114
|
+
"%s: %d/%d (%d%%)", label, count, total, count * 100 // total
|
|
115
|
+
)
|
|
116
|
+
else:
|
|
117
|
+
logger.info("%s: %d done", label, count)
|
|
118
|
+
# Final line unless the last item already landed on a logged boundary.
|
|
119
|
+
if count and count % log_every != 0:
|
|
120
|
+
if total:
|
|
121
|
+
logger.info("%s: %d/%d (100%%)", label, count, total)
|
|
122
|
+
else:
|
|
123
|
+
logger.info("%s: %d done", label, count)
|
|
124
|
+
return
|
|
125
|
+
|
|
126
|
+
yield from iterable
|
|
@@ -1,23 +0,0 @@
|
|
|
1
|
-
"""General-purpose utilities."""
|
|
2
|
-
|
|
3
|
-
from typing import Iterator
|
|
4
|
-
|
|
5
|
-
import pandas as pd
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
def chunker(df: pd.DataFrame, chunksize: int = 10000) -> Iterator[pd.DataFrame]:
|
|
9
|
-
"""Yield successive non-overlapping chunks of *df*.
|
|
10
|
-
|
|
11
|
-
Parameters
|
|
12
|
-
----------
|
|
13
|
-
df : pd.DataFrame
|
|
14
|
-
The DataFrame to split.
|
|
15
|
-
chunksize : int
|
|
16
|
-
Number of rows per chunk (default 10 000).
|
|
17
|
-
|
|
18
|
-
Yields
|
|
19
|
-
------
|
|
20
|
-
pd.DataFrame
|
|
21
|
-
"""
|
|
22
|
-
for start in range(0, len(df), chunksize):
|
|
23
|
-
yield df.iloc[start : start + chunksize]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|