eosframes 1.1.0__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: eosframes
3
- Version: 1.1.0
3
+ Version: 1.2.0
4
4
  Summary: Ersilia utilities for working with tabular output data
5
5
  License: MIT
6
6
  License-File: LICENSE
@@ -19,6 +19,7 @@ Classifier: Programming Language :: Python :: 3.11
19
19
  Classifier: Programming Language :: Python :: 3.12
20
20
  Classifier: Programming Language :: Python :: 3.13
21
21
  Classifier: Programming Language :: Python :: 3.14
22
+ Classifier: Programming Language :: Python :: 3.15
22
23
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
24
  Classifier: Topic :: Scientific/Engineering :: Chemistry
24
25
  Requires-Dist: click (>=8.0)
@@ -98,11 +99,41 @@ Run `eosframes --help` (or `eosframes <command> --help`) for inline help.
98
99
 
99
100
  See [`docs/cli.md`](docs/cli.md) for every flag, example, and refusal condition.
100
101
 
102
+ `fit` and `transform` stream in bounded memory — `fit` walks one column at a
103
+ time, `transform` one row-chunk at a time — so they handle files far larger
104
+ than RAM (wide fingerprint frames, tens of GB). Tune `--chunksize` to trade
105
+ memory for throughput.
106
+
107
+ ## Scripts
108
+
109
+ [`scripts/build_scaler.sh`](scripts/build_scaler.sh) builds and packages a scaler
110
+ for one model in a single step: it pulls the model's precalculations from the
111
+ [isaura](https://github.com/ersilia-os/isaura) store over the Ersilia reference
112
+ library (`data/ersilia_reference_library_v0.csv`), fits an `eosframes` scaler, and
113
+ compresses the transformer into a versioned zip.
114
+
115
+ ```bash
116
+ scripts/build_scaler.sh <model_id> <version> # e.g. eos4e40 v1
117
+ ```
118
+
119
+ The artifact is written to:
120
+
121
+ ```
122
+ output/ersilia_reference_library_v0/<model_id>/<version>/scaler-<eosframes-major>.zip
123
+ ```
124
+
125
+ (e.g. `output/ersilia_reference_library_v0/eos4e40/v1/scaler-1.zip`), containing a
126
+ single `<model_id>_<version>_transformer.json`. The bucket defaults to
127
+ `isaura-public`; override it with the `PROJECT_NAME` environment variable.
128
+
129
+ **Prerequisites:** `isaura` and `eosframes` on `PATH`, and a running local isaura
130
+ MinIO engine (`isaura engine --start`).
131
+
101
132
  ## Documentation
102
133
 
103
134
  - [`docs/cli.md`](docs/cli.md) — every CLI command, all flags, examples, and error patterns.
104
135
  - [`docs/nomenclature.md`](docs/nomenclature.md) — every recognised filename / directory pattern, the strict/lenient contract, and the two stack modes.
105
- - [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, and quantization / imputation.
136
+ - [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, quantization / imputation, and the bounded-memory streaming fit/transform for large files.
106
137
 
107
138
  ## About the Ersilia Open Source Initiative
108
139
 
@@ -65,11 +65,41 @@ Run `eosframes --help` (or `eosframes <command> --help`) for inline help.
65
65
 
66
66
  See [`docs/cli.md`](docs/cli.md) for every flag, example, and refusal condition.
67
67
 
68
+ `fit` and `transform` stream in bounded memory — `fit` walks one column at a
69
+ time, `transform` one row-chunk at a time — so they handle files far larger
70
+ than RAM (wide fingerprint frames, tens of GB). Tune `--chunksize` to trade
71
+ memory for throughput.
72
+
73
+ ## Scripts
74
+
75
+ [`scripts/build_scaler.sh`](scripts/build_scaler.sh) builds and packages a scaler
76
+ for one model in a single step: it pulls the model's precalculations from the
77
+ [isaura](https://github.com/ersilia-os/isaura) store over the Ersilia reference
78
+ library (`data/ersilia_reference_library_v0.csv`), fits an `eosframes` scaler, and
79
+ compresses the transformer into a versioned zip.
80
+
81
+ ```bash
82
+ scripts/build_scaler.sh <model_id> <version> # e.g. eos4e40 v1
83
+ ```
84
+
85
+ The artifact is written to:
86
+
87
+ ```
88
+ output/ersilia_reference_library_v0/<model_id>/<version>/scaler-<eosframes-major>.zip
89
+ ```
90
+
91
+ (e.g. `output/ersilia_reference_library_v0/eos4e40/v1/scaler-1.zip`), containing a
92
+ single `<model_id>_<version>_transformer.json`. The bucket defaults to
93
+ `isaura-public`; override it with the `PROJECT_NAME` environment variable.
94
+
95
+ **Prerequisites:** `isaura` and `eosframes` on `PATH`, and a running local isaura
96
+ MinIO engine (`isaura engine --start`).
97
+
68
98
  ## Documentation
69
99
 
70
100
  - [`docs/cli.md`](docs/cli.md) — every CLI command, all flags, examples, and error patterns.
71
101
  - [`docs/nomenclature.md`](docs/nomenclature.md) — every recognised filename / directory pattern, the strict/lenient contract, and the two stack modes.
72
- - [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, and quantization / imputation.
102
+ - [`docs/scaling.md`](docs/scaling.md) — the type-aware robust scaler: column kinds, how each is picked, quantization / imputation, and the bounded-memory streaming fit/transform for large files.
73
103
 
74
104
  ## About the Ersilia Open Source Initiative
75
105
 
@@ -9,7 +9,7 @@ style = "pep440"
9
9
 
10
10
  [tool.poetry]
11
11
  name = "eosframes"
12
- version = "1.1.0"
12
+ version = "1.2.0"
13
13
  description = "Ersilia utilities for working with tabular output data"
14
14
  authors = ["Ersilia Open Source Initiative <hello@ersilia.io>"]
15
15
  license = "MIT"
@@ -186,9 +186,7 @@ def _render_sidecar(
186
186
  click.echo(resolved_out)
187
187
 
188
188
 
189
- def _fetch_or_clickfail(
190
- fetch: Callable, *args, **kwargs
191
- ):
189
+ def _fetch_or_clickfail(fetch: Callable, *args, **kwargs):
192
190
  """Call a hub fetcher and convert ``EosframesError`` to ``ClickException``."""
193
191
  try:
194
192
  return fetch(*args, **kwargs)
@@ -664,7 +662,27 @@ def columns(input_file: str, output: str) -> None:
664
662
  "transform time."
665
663
  ),
666
664
  )
667
- def fit(input_file: str, scaler: str, output: str, quantize: bool, impute: bool) -> None:
665
+ @click.option(
666
+ "--chunksize",
667
+ "chunksize",
668
+ type=int,
669
+ default=_scale._DEFAULT_CHUNKSIZE,
670
+ show_default=True,
671
+ help=(
672
+ "Rows per streamed chunk. The fit never holds the whole file — it "
673
+ "walks one column at a time (CSV inputs are first staged to a "
674
+ "temporary columnar H5 in chunks of this size). Lower it for very "
675
+ "wide frames, raise it for narrow ones."
676
+ ),
677
+ )
678
+ def fit(
679
+ input_file: str,
680
+ scaler: str,
681
+ output: str,
682
+ quantize: bool,
683
+ impute: bool,
684
+ chunksize: int,
685
+ ) -> None:
668
686
  """Fit a type-aware robust scaler on INPUT_FILE and save parameters to SCALER.
669
687
 
670
688
  Each numeric feature column is auto-classified (constant, binary, count,
@@ -700,6 +718,7 @@ def fit(input_file: str, scaler: str, output: str, quantize: bool, impute: bool)
700
718
  output_path=output,
701
719
  output_dtype=output_dtype,
702
720
  impute=impute,
721
+ chunksize=chunksize,
703
722
  )
704
723
  except EosframesError as e:
705
724
  raise _err(e) from e
@@ -744,7 +763,26 @@ def fit(input_file: str, scaler: str, output: str, quantize: bool, impute: bool)
744
763
  "no -128 sentinels)."
745
764
  ),
746
765
  )
747
- def transform(input_file: str, scaler: str, output: str, quantize: bool, impute: bool) -> None:
766
+ @click.option(
767
+ "--chunksize",
768
+ "chunksize",
769
+ type=int,
770
+ default=_scale._DEFAULT_CHUNKSIZE,
771
+ show_default=True,
772
+ help=(
773
+ "Rows per streamed chunk. The input is read and the output written "
774
+ "one chunk at a time, so memory stays bounded regardless of file "
775
+ "size. Lower it for very wide frames, raise it for narrow ones."
776
+ ),
777
+ )
778
+ def transform(
779
+ input_file: str,
780
+ scaler: str,
781
+ output: str,
782
+ quantize: bool,
783
+ impute: bool,
784
+ chunksize: int,
785
+ ) -> None:
748
786
  """Apply a saved scaler to INPUT_FILE and write scaled data to OUTPUT.
749
787
 
750
788
  Loads the scaler parameters from SCALER and applies them to the numeric
@@ -757,7 +795,12 @@ def transform(input_file: str, scaler: str, output: str, quantize: bool, impute:
757
795
  output_dtype = "int8" if quantize else "float32"
758
796
  try:
759
797
  out = _scale.transform_file(
760
- input_file, scaler, output, output_dtype=output_dtype, impute=impute
798
+ input_file,
799
+ scaler,
800
+ output,
801
+ output_dtype=output_dtype,
802
+ impute=impute,
803
+ chunksize=chunksize,
761
804
  )
762
805
  except EosframesError as e:
763
806
  raise _err(e) from e
@@ -76,9 +76,7 @@ def _require_no_overwrite(path: str, *, kind: str = "file") -> None:
76
76
  f"Output folder '{path}' already exists. "
77
77
  "Remove it or choose a different name."
78
78
  )
79
- raise EosframesError(
80
- f"Output file '{path}' already exists. Remove it first."
81
- )
79
+ raise EosframesError(f"Output file '{path}' already exists. Remove it first.")
82
80
 
83
81
 
84
82
  def _compute_summary_stats(df: pd.DataFrame) -> List[Dict]:
@@ -173,8 +173,9 @@ hand-maintained schema number.
173
173
 
174
174
  import json
175
175
  import os
176
+ import tempfile
176
177
  from datetime import datetime
177
- from typing import Callable, Dict, Optional, Tuple
178
+ from typing import Callable, Dict, Iterator, Optional, Tuple
178
179
 
179
180
  import h5py
180
181
  import numpy as np
@@ -189,6 +190,7 @@ from .naming import (
189
190
  parse_name,
190
191
  parse_transformer_name,
191
192
  )
193
+ from .utils import progress
192
194
 
193
195
  _META_COLS = {"key", "input"}
194
196
 
@@ -197,6 +199,31 @@ _METHOD_NAME = "robust_typed"
197
199
  _VALID_OUTPUT_DTYPES = ("float32", "int8")
198
200
  _DEFAULT_OUTPUT_DTYPE = "float32"
199
201
 
202
+ # Rows per chunk for the streaming file-level fit/transform paths. Big
203
+ # enough to amortise per-chunk overhead, small enough that one chunk of a
204
+ # wide frame stays well under a GB (e.g. 50k rows × 3k cols × 4B ≈ 0.6 GB).
205
+ _DEFAULT_CHUNKSIZE = 50_000
206
+
207
+ # Row cadence for the CSV→H5 staging milestone log (independent of chunksize,
208
+ # so the log reads the same whatever chunk size the caller picks).
209
+ _STAGE_LOG_ROWS = 1_000_000
210
+
211
+ # Columns pulled per HDF5 read during the fit. Reading the values dataset one
212
+ # column at a time forces a re-read of every storage chunk the column touches,
213
+ # so we read a batch of columns at once and then fit each column from the
214
+ # in-memory block. 100 cols × n_rows × 4 B stays bounded (~0.5 GB at 1.3 M
215
+ # rows) while cutting the number of HDF5 reads ~100×.
216
+ _FIT_COLUMN_BATCH = 100
217
+
218
+ # Target bytes per HDF5 storage chunk for the staged values dataset. Chunks are
219
+ # tiled ``(row_tile, _FIT_COLUMN_BATCH)`` so the fit's column-batch reads align
220
+ # to chunk boundaries — each chunk is read exactly once across the whole fit.
221
+ _STAGE_CHUNK_BYTES = 4 * 1024 * 1024
222
+
223
+ # Target number of progress log lines for the column-fit and row-chunk
224
+ # transform loops when running off-TTY (one line per ~5% of the work).
225
+ _PROGRESS_LOG_STEPS = 20
226
+
200
227
  _INT8_NAN_SENTINEL = -128
201
228
  _INT8_MAX_VAL = 127
202
229
 
@@ -309,9 +336,7 @@ _COUNT_SHIFTED_EXTENT_RATIO_MAX = 2.0
309
336
  _PIECEWISE_BLEND_FRACTION = 0.5
310
337
 
311
338
 
312
- def _tanh_tail(
313
- mag: np.ndarray, body_extent: float, body_target: float
314
- ) -> np.ndarray:
339
+ def _tanh_tail(mag: np.ndarray, body_extent: float, body_target: float) -> np.ndarray:
315
340
  """Asymptotic tail for ``|x - center| > body_extent``.
316
341
 
317
342
  Returns ``y = body_target + (1 − body_target)·tanh(c·u)`` for
@@ -596,12 +621,10 @@ def _fit_continuous(series: pd.Series) -> dict:
596
621
  tail_asymmetry_right = right_span / max(left_span, _eps)
597
622
  tail_asymmetry_left = left_span / max(right_span, _eps)
598
623
  is_right_skew = (
599
- bowley > _BOWLEY_THRESHOLD
600
- and tail_asymmetry_right > _TAIL_ASYMMETRY_THRESHOLD
624
+ bowley > _BOWLEY_THRESHOLD and tail_asymmetry_right > _TAIL_ASYMMETRY_THRESHOLD
601
625
  )
602
626
  is_left_skew = (
603
- bowley < -_BOWLEY_THRESHOLD
604
- and tail_asymmetry_left > _TAIL_ASYMMETRY_THRESHOLD
627
+ bowley < -_BOWLEY_THRESHOLD and tail_asymmetry_left > _TAIL_ASYMMETRY_THRESHOLD
605
628
  )
606
629
 
607
630
  if is_right_skew:
@@ -756,6 +779,9 @@ def _fit_count(series: pd.Series) -> dict:
756
779
  # Flag near-degenerate sparse counts where most rows are 0 and
757
780
  # the scaler can only produce a handful of distinct values. Goes
758
781
  # into fit_notes — advisory only, never read by the transform.
782
+ # No per-column log here: sparse fingerprints (e.g. Morgan counts)
783
+ # flag hundreds of columns and would flood the log. fit() emits a
784
+ # single aggregated summary instead.
759
785
  scaled = np.clip(arr / high_anchor, 0.0, 1.0)
760
786
  n_distinct = int(np.unique(scaled).size)
761
787
  mode_fraction = float((arr == 0.0).sum()) / float(arr.size)
@@ -768,15 +794,6 @@ def _fit_count(series: pd.Series) -> dict:
768
794
  "mode_fraction": mode_fraction,
769
795
  "n_distinct_output": n_distinct,
770
796
  }
771
- get_logger().warning(
772
- "Count column '%s' is near-degenerate "
773
- "(mode_fraction=%.2f, n_distinct_output=%d). "
774
- "Output collapses to a handful of values — "
775
- "consider dropping it or revisiting upstream featurization.",
776
- series.name,
777
- mode_fraction,
778
- n_distinct,
779
- )
780
797
  return entry
781
798
 
782
799
  # Count with a non-zero mode. Linear+clip on each side of the mode,
@@ -889,9 +906,7 @@ def _apply_constant(
889
906
  return _to_output_array(out, output_dtype)
890
907
 
891
908
 
892
- def _apply_binary(
893
- series: pd.Series, transform: dict, output_dtype: str
894
- ) -> np.ndarray:
909
+ def _apply_binary(series: pd.Series, transform: dict, output_dtype: str) -> np.ndarray:
895
910
  low = float(transform["low"])
896
911
  high = float(transform["high"])
897
912
  arr = series.to_numpy(dtype=float)
@@ -1123,9 +1138,7 @@ def _apply_continuous_left(
1123
1138
  else: # "finite"
1124
1139
  low = float(transform["low_anchor"])
1125
1140
  max_extent = high - low
1126
- out[tail_mask] = -_quadratic_tail(
1127
- mag, body_span, body_target, max_extent
1128
- )
1141
+ out[tail_mask] = -_quadratic_tail(mag, body_span, body_target, max_extent)
1129
1142
 
1130
1143
  # Cubic Hermite blend across the body→middle junction (mirror of
1131
1144
  # the right-skew blend).
@@ -1193,7 +1206,9 @@ def _apply_continuous_centered(
1193
1206
  nan = np.isnan(arr)
1194
1207
  out = np.full(arr.shape, np.nan, dtype=float)
1195
1208
 
1196
- def _side(mask: np.ndarray, body_extent: float, max_extent: float, sign: float) -> None:
1209
+ def _side(
1210
+ mask: np.ndarray, body_extent: float, max_extent: float, sign: float
1211
+ ) -> None:
1197
1212
  if not mask.any():
1198
1213
  return
1199
1214
  if body_extent <= 0:
@@ -1249,6 +1264,73 @@ def _dispatch_apply(
1249
1264
  # ---------------------------------------------------------------------------
1250
1265
 
1251
1266
 
1267
+ def _fit_series(series: pd.Series) -> dict:
1268
+ """Classify one column and fit its type-specific transform.
1269
+
1270
+ Shared by the in-memory :func:`fit` and the streaming
1271
+ :func:`fit_file`. All-NaN columns and columns that slip past
1272
+ classification without a usable scale both fall back to
1273
+ ``kind: "constant"``.
1274
+ """
1275
+ if series.dropna().empty:
1276
+ # All-NaN column: fit as a constant. The dispatch at transform
1277
+ # time maps non-NaN inputs to 0 and propagates NaN.
1278
+ return _fit_constant(series)
1279
+ type_ = _classify_type(series)
1280
+ if type_ == "constant":
1281
+ return _fit_constant(series)
1282
+ if type_ == "binary":
1283
+ return _fit_binary(series)
1284
+ if type_ == "count":
1285
+ return _fit_count(series)
1286
+ try:
1287
+ return _fit_continuous(series)
1288
+ except EosframesError:
1289
+ # Column slipped past _classify_type but has no usable scale.
1290
+ return _fit_constant(series)
1291
+
1292
+
1293
+ def _log_fit_summary(columns: dict, logger) -> None:
1294
+ """Emit the per-fit kind breakdown and near-degenerate advisory.
1295
+
1296
+ Shared by :func:`fit` and :func:`fit_file` so the streaming and
1297
+ in-memory paths log identically.
1298
+ """
1299
+ kind_counts: dict = {}
1300
+ for entry in columns.values():
1301
+ kind = entry["transform"]["kind"]
1302
+ kind_counts[kind] = kind_counts.get(kind, 0) + 1
1303
+
1304
+ kind_breakdown = ", ".join(
1305
+ f"{kind}={count}" for kind, count in sorted(kind_counts.items())
1306
+ )
1307
+ n = len(columns)
1308
+ logger.info("Fitted %d / %d numeric columns (%s)", n, n, kind_breakdown)
1309
+
1310
+ # Single aggregated advisory for near-degenerate count columns. The
1311
+ # per-column flag lives in each entry's fit_notes; here we just
1312
+ # summarise how many tripped it so sparse fingerprints don't flood
1313
+ # the log with one warning per bit.
1314
+ degenerate_cols = [
1315
+ col
1316
+ for col, entry in columns.items()
1317
+ if entry.get("fit_notes", {}).get("degenerate")
1318
+ ]
1319
+ if degenerate_cols:
1320
+ preview = ", ".join(str(c) for c in degenerate_cols[:5])
1321
+ if len(degenerate_cols) > 5:
1322
+ preview += ", …"
1323
+ logger.warning(
1324
+ "%d of %d count columns are near-degenerate (e.g. %s): output "
1325
+ "collapses to a handful of values. Common for sparse "
1326
+ "fingerprints; see each column's fit_notes for details. "
1327
+ "Consider dropping them or revisiting upstream featurization.",
1328
+ len(degenerate_cols),
1329
+ n,
1330
+ preview,
1331
+ )
1332
+
1333
+
1252
1334
  def fit(df: pd.DataFrame) -> dict:
1253
1335
  """Fit a type-aware robust scaler on the numeric feature columns.
1254
1336
 
@@ -1294,44 +1376,10 @@ def fit(df: pd.DataFrame) -> dict:
1294
1376
  raise EosframesError("No numeric feature columns found to fit the scaler.")
1295
1377
 
1296
1378
  columns: dict = {}
1297
- kind_counts: dict = {}
1298
-
1299
- for col in numeric_cols:
1300
- series = df[col]
1301
-
1302
- if series.dropna().empty:
1303
- # All-NaN column: fit as a constant. The dispatch at
1304
- # transform time maps non-NaN inputs to 0 and propagates NaN.
1305
- entry = _fit_constant(series)
1306
- else:
1307
- type_ = _classify_type(series)
1308
- if type_ == "constant":
1309
- entry = _fit_constant(series)
1310
- elif type_ == "binary":
1311
- entry = _fit_binary(series)
1312
- elif type_ == "count":
1313
- entry = _fit_count(series)
1314
- else:
1315
- try:
1316
- entry = _fit_continuous(series)
1317
- except EosframesError:
1318
- # Column slipped past _classify_type but has no usable scale.
1319
- entry = _fit_constant(series)
1320
-
1321
- columns[col] = entry
1322
- kind = entry["transform"]["kind"]
1323
- kind_counts[kind] = kind_counts.get(kind, 0) + 1
1324
-
1325
- kind_breakdown = ", ".join(
1326
- f"{kind}={count}" for kind, count in sorted(kind_counts.items())
1327
- )
1328
- logger.info(
1329
- "Fitted %d / %d numeric columns (%s)",
1330
- len(numeric_cols),
1331
- len(numeric_cols),
1332
- kind_breakdown,
1333
- )
1379
+ for col in progress(numeric_cols, desc="Fitting columns"):
1380
+ columns[col] = _fit_series(df[col])
1334
1381
 
1382
+ _log_fit_summary(columns, logger)
1335
1383
  return {"method": _METHOD_NAME, "columns": columns}
1336
1384
 
1337
1385
 
@@ -1381,10 +1429,15 @@ def transform(
1381
1429
  On column mismatch, invalid ``output_dtype``, or unknown
1382
1430
  column type in *params*.
1383
1431
  """
1432
+ _validate_transform(df, params, output_dtype)
1433
+ return _apply_transform(df, params, output_dtype, impute, show_progress=True)
1434
+
1435
+
1436
+ def _validate_transform(df: pd.DataFrame, params: dict, output_dtype: str) -> None:
1437
+ """Check output dtype and exact feature-column match before applying."""
1384
1438
  if output_dtype not in _VALID_OUTPUT_DTYPES:
1385
1439
  raise EosframesError(
1386
- f"Unknown output_dtype '{output_dtype}'. "
1387
- f"Supported: {_VALID_OUTPUT_DTYPES}"
1440
+ f"Unknown output_dtype '{output_dtype}'. Supported: {_VALID_OUTPUT_DTYPES}"
1388
1441
  )
1389
1442
 
1390
1443
  expected_feature_cols = list(params["columns"].keys())
@@ -1396,9 +1449,29 @@ def transform(
1396
1449
  f"but transformer was fitted on {expected_feature_cols}."
1397
1450
  )
1398
1451
 
1452
+
1453
+ def _apply_transform(
1454
+ df: pd.DataFrame,
1455
+ params: dict,
1456
+ output_dtype: str,
1457
+ impute: bool,
1458
+ show_progress: bool,
1459
+ ) -> pd.DataFrame:
1460
+ """Apply the fitted per-column transforms to *df* (already validated).
1461
+
1462
+ *show_progress* draws the per-column bar — on for the single-shot
1463
+ in-memory :func:`transform`, off for the streaming path where the bar
1464
+ would otherwise redraw once per row-chunk.
1465
+ """
1399
1466
  columns = params["columns"]
1467
+ expected_feature_cols = list(columns.keys())
1400
1468
  result = df.copy()
1401
- for col in expected_feature_cols:
1469
+ iterator = (
1470
+ progress(expected_feature_cols, desc="Transforming columns")
1471
+ if show_progress
1472
+ else expected_feature_cols
1473
+ )
1474
+ for col in iterator:
1402
1475
  entry = columns[col]
1403
1476
  series = df[col]
1404
1477
  if impute:
@@ -1418,32 +1491,260 @@ def _values_dtype_for(output_dtype: str) -> np.dtype:
1418
1491
  return np.float32
1419
1492
 
1420
1493
 
1421
- def _write_df(
1422
- df: pd.DataFrame, output_path: str, values_dtype: np.dtype = np.float32
1423
- ) -> None:
1424
- """Write a DataFrame to CSV or H5, bypassing the naming convention check.
1494
+ # ---------------------------------------------------------------------------
1495
+ # Streaming primitives — bounded-memory fit/transform for very wide or very
1496
+ # tall files. The whole matrix is never resident: fit walks one column at a
1497
+ # time (cheap on the columnar H5 layout), transform walks row-chunks.
1498
+ # ---------------------------------------------------------------------------
1499
+
1500
+
1501
+ def _h5_feature_names(h5_path: str) -> list:
1502
+ with h5py.File(h5_path, "r") as f:
1503
+ return [x.decode("utf-8") for x in f["features"][:]]
1504
+
1505
+
1506
+ def _h5_nrows(h5_path: str) -> int:
1507
+ with h5py.File(h5_path, "r") as f:
1508
+ return int(f["values"].shape[0])
1509
+
1510
+
1511
+ def _iter_h5_columns(
1512
+ h5_path: str, batch: int = _FIT_COLUMN_BATCH
1513
+ ) -> Iterator[Tuple[str, np.ndarray]]:
1514
+ """Yield ``(feature_name, full_column_array)`` for every feature column.
1515
+
1516
+ Columns are pulled from disk a *batch* at a time — one ``values[:,
1517
+ start:stop]`` hyperslab — then handed out one at a time from that
1518
+ in-memory block. This keeps peak memory bounded (``batch`` columns, not
1519
+ the whole matrix) while reading each underlying HDF5 storage chunk once,
1520
+ instead of re-reading it for every column it contains.
1521
+ """
1522
+ with h5py.File(h5_path, "r") as f:
1523
+ features = [x.decode("utf-8") for x in f["features"][:]]
1524
+ values = f["values"]
1525
+ n_features = len(features)
1526
+ for start in range(0, n_features, batch):
1527
+ block = values[:, start : start + batch] # (n_rows, ≤batch)
1528
+ for k in range(block.shape[1]):
1529
+ yield features[start + k], block[:, k]
1530
+
1531
+
1532
+ def _stage_csv_to_h5(csv_path: str, h5_path: str, chunksize: int) -> Tuple[list, int]:
1533
+ """Stream a CSV into a columnar temp H5 in one bounded-memory pass.
1534
+
1535
+ Reads the CSV in row-chunks and appends each chunk's numeric feature
1536
+ columns to a resizable ``values`` dataset (``float32`` — matches the
1537
+ pipeline's default output precision). Returns ``(feature_cols, n_rows)``.
1538
+ The resulting H5 is then cheaply column-sliceable for the fit.
1539
+ """
1540
+ logger = get_logger()
1541
+ feature_cols: Optional[list] = None
1542
+ total = 0
1543
+ next_log = _STAGE_LOG_ROWS
1544
+ with h5py.File(h5_path, "w") as f:
1545
+ values_ds = None
1546
+ for chunk in pd.read_csv(csv_path, chunksize=chunksize):
1547
+ if feature_cols is None:
1548
+ feat = [c for c in chunk.columns if c not in _META_COLS]
1549
+ feature_cols = [
1550
+ c for c in feat if pd.api.types.is_numeric_dtype(chunk[c])
1551
+ ]
1552
+ n_feat = len(feature_cols)
1553
+ # Tile chunks to the fit's column-batch width so each storage
1554
+ # chunk is read exactly once during the (batched) column fit.
1555
+ if n_feat > 0:
1556
+ col_tile = min(n_feat, _FIT_COLUMN_BATCH)
1557
+ row_tile = max(1, _STAGE_CHUNK_BYTES // (col_tile * 4))
1558
+ chunk_shape: object = (row_tile, col_tile)
1559
+ else:
1560
+ chunk_shape = True
1561
+ values_ds = f.create_dataset(
1562
+ "values",
1563
+ shape=(0, n_feat),
1564
+ maxshape=(None, n_feat),
1565
+ dtype=np.float32,
1566
+ chunks=chunk_shape,
1567
+ )
1568
+ logger.info("Staging: detected %d numeric feature columns", n_feat)
1569
+ arr = chunk[feature_cols].to_numpy(dtype=np.float32)
1570
+ n = arr.shape[0]
1571
+ values_ds.resize(total + n, axis=0)
1572
+ values_ds[total : total + n, :] = arr
1573
+ total += n
1574
+ if total >= next_log:
1575
+ logger.info("Staging: %s rows written to temp H5", f"{total:,}")
1576
+ next_log += _STAGE_LOG_ROWS
1577
+ dt = h5py.string_dtype(encoding="utf-8")
1578
+ f.create_dataset("features", data=feature_cols or [], dtype=dt)
1579
+ logger.info(
1580
+ "Staging complete: %s rows × %d columns → %s",
1581
+ f"{total:,}",
1582
+ len(feature_cols or []),
1583
+ os.path.basename(h5_path),
1584
+ )
1585
+ return feature_cols or [], total
1586
+
1587
+
1588
+ def _iter_row_chunks(path: str, chunksize: int) -> Iterator[pd.DataFrame]:
1589
+ """Yield row-chunk DataFrames (``key``/``input`` + feature cols) from a file.
1425
1590
 
1426
- *values_dtype* controls the H5 ``values`` dataset dtype; CSV ignores it.
1591
+ CSV is streamed natively by pandas; H5 is sliced ``[start:end, :]`` so
1592
+ only ``chunksize`` rows are resident at a time. The yielded frames have
1593
+ the same column layout the in-memory :func:`transform` expects.
1427
1594
  """
1428
- ext = os.path.splitext(output_path)[1].lower()
1595
+ ext = os.path.splitext(path)[1].lower()
1429
1596
  if ext == ".csv":
1430
- df.to_csv(output_path, index=False)
1597
+ yield from pd.read_csv(path, chunksize=chunksize)
1431
1598
  elif ext == ".h5":
1432
- feat_cols = [c for c in df.columns if c not in _META_COLS]
1433
- with h5py.File(output_path, "w") as f:
1434
- dt = h5py.string_dtype(encoding="utf-8")
1435
- if "key" in df.columns:
1436
- f.create_dataset("key", data=df["key"].astype(str).tolist(), dtype=dt)
1437
- if "input" in df.columns:
1438
- f.create_dataset(
1439
- "input", data=df["input"].astype(str).tolist(), dtype=dt
1440
- )
1441
- f.create_dataset("features", data=feat_cols, dtype=dt)
1442
- f.create_dataset(
1443
- "values", data=df[feat_cols].values, dtype=values_dtype
1444
- )
1599
+ with h5py.File(path, "r") as f:
1600
+ features = [x.decode("utf-8") for x in f["features"][:]]
1601
+ values = f["values"]
1602
+ n_rows = values.shape[0]
1603
+ has_key = "key" in f
1604
+ for start in range(0, n_rows, chunksize):
1605
+ end = min(start + chunksize, n_rows)
1606
+ meta = {}
1607
+ if has_key:
1608
+ meta["key"] = [x.decode("utf-8") for x in f["key"][start:end]]
1609
+ meta["input"] = [x.decode("utf-8") for x in f["input"][start:end]]
1610
+ block = pd.DataFrame(values[start:end, :], columns=features)
1611
+ yield pd.concat([pd.DataFrame(meta), block], axis=1)
1445
1612
  else:
1446
- raise EosframesError(f"Unsupported output format '{ext}'. Expected .csv or .h5")
1613
+ raise EosframesError(f"Unsupported input format '{ext}'. Expected .csv or .h5")
1614
+
1615
+
1616
+ class _StreamWriter:
1617
+ """Incremental CSV/H5 writer — appends row-chunks without holding the file.
1618
+
1619
+ CSV: header on the first chunk, append thereafter. H5: resizable
1620
+ ``values`` / ``input`` / ``key`` datasets created from the first chunk's
1621
+ schema, then extended per chunk. Bypasses the naming-convention check
1622
+ (the caller has already validated the output path).
1623
+ """
1624
+
1625
+ def __init__(self, path: str, values_dtype: np.dtype):
1626
+ self.path = path
1627
+ self.ext = os.path.splitext(path)[1].lower()
1628
+ self.values_dtype = values_dtype
1629
+ if self.ext not in (".csv", ".h5"):
1630
+ raise EosframesError(
1631
+ f"Unsupported output format '{self.ext}'. Expected .csv or .h5"
1632
+ )
1633
+ self._csv_started = False
1634
+ self._h5 = None
1635
+ self._feature_cols: Optional[list] = None
1636
+ self._has_key = False
1637
+ self._n = 0
1638
+
1639
+ def write(self, df: pd.DataFrame) -> None:
1640
+ if self.ext == ".csv":
1641
+ df.to_csv(
1642
+ self.path,
1643
+ mode="w" if not self._csv_started else "a",
1644
+ header=not self._csv_started,
1645
+ index=False,
1646
+ )
1647
+ self._csv_started = True
1648
+ else:
1649
+ if self._h5 is None:
1650
+ self._init_h5(df)
1651
+ self._append_h5(df)
1652
+
1653
+ def _init_h5(self, df: pd.DataFrame) -> None:
1654
+ self._feature_cols = [c for c in df.columns if c not in _META_COLS]
1655
+ self._has_key = "key" in df.columns
1656
+ self._h5 = h5py.File(self.path, "w")
1657
+ dt = h5py.string_dtype(encoding="utf-8")
1658
+ self._h5.create_dataset("features", data=self._feature_cols, dtype=dt)
1659
+ self._values = self._h5.create_dataset(
1660
+ "values",
1661
+ shape=(0, len(self._feature_cols)),
1662
+ maxshape=(None, len(self._feature_cols)),
1663
+ dtype=self.values_dtype,
1664
+ chunks=True,
1665
+ )
1666
+ self._input = self._h5.create_dataset(
1667
+ "input", shape=(0,), maxshape=(None,), dtype=dt
1668
+ )
1669
+ if self._has_key:
1670
+ self._key = self._h5.create_dataset(
1671
+ "key", shape=(0,), maxshape=(None,), dtype=dt
1672
+ )
1673
+
1674
+ def _append_h5(self, df: pd.DataFrame) -> None:
1675
+ n = len(df)
1676
+ new = self._n + n
1677
+ self._values.resize(new, axis=0)
1678
+ self._values[self._n : new, :] = df[self._feature_cols].to_numpy(
1679
+ dtype=self.values_dtype
1680
+ )
1681
+ self._input.resize(new, axis=0)
1682
+ self._input[self._n : new] = df["input"].astype(str).tolist()
1683
+ if self._has_key:
1684
+ self._key.resize(new, axis=0)
1685
+ self._key[self._n : new] = df["key"].astype(str).tolist()
1686
+ self._n = new
1687
+
1688
+ def close(self) -> None:
1689
+ if self._h5 is not None:
1690
+ self._h5.close()
1691
+ self._h5 = None
1692
+
1693
+
1694
+ def _streaming_transform(
1695
+ input_path: str,
1696
+ transformer: dict,
1697
+ output_path: str,
1698
+ output_dtype: str,
1699
+ impute: bool,
1700
+ chunksize: int,
1701
+ ) -> int:
1702
+ """Row-chunked transform: read a chunk, scale it, write it, repeat.
1703
+
1704
+ Reuses the in-memory per-column dispatch on each small chunk (with its
1705
+ per-column bar suppressed), so peak memory is one chunk in + one chunk
1706
+ out — independent of file size. Returns the number of rows written.
1707
+ """
1708
+ logger = get_logger()
1709
+ in_ext = os.path.splitext(input_path)[1].lower()
1710
+ # Row count is free for H5 (so we get a true % bar / steady log cadence);
1711
+ # for CSV it is unknown without a full scan, so we log every fixed number
1712
+ # of chunks instead.
1713
+ total_chunks = None
1714
+ if in_ext == ".h5":
1715
+ total_chunks = (_h5_nrows(input_path) + chunksize - 1) // chunksize
1716
+ logger.info(
1717
+ "Transforming %s rows in %d chunks of %d",
1718
+ f"{_h5_nrows(input_path):,}",
1719
+ total_chunks,
1720
+ chunksize,
1721
+ )
1722
+ log_every = (
1723
+ max(1, total_chunks // _PROGRESS_LOG_STEPS)
1724
+ if total_chunks
1725
+ else _PROGRESS_LOG_STEPS
1726
+ )
1727
+
1728
+ writer = _StreamWriter(output_path, _values_dtype_for(output_dtype))
1729
+ chunks = progress(
1730
+ _iter_row_chunks(input_path, chunksize),
1731
+ total=total_chunks,
1732
+ desc="Transforming chunks",
1733
+ log_every=log_every,
1734
+ )
1735
+
1736
+ total_rows = 0
1737
+ try:
1738
+ for chunk in chunks:
1739
+ _validate_transform(chunk, transformer, output_dtype)
1740
+ scaled = _apply_transform(
1741
+ chunk, transformer, output_dtype, impute, show_progress=False
1742
+ )
1743
+ writer.write(scaled)
1744
+ total_rows += len(scaled)
1745
+ finally:
1746
+ writer.close()
1747
+ return total_rows
1447
1748
 
1448
1749
 
1449
1750
  def fit_file(
@@ -1452,15 +1753,27 @@ def fit_file(
1452
1753
  output_path: Optional[str] = None,
1453
1754
  output_dtype: str = _DEFAULT_OUTPUT_DTYPE,
1454
1755
  impute: bool = False,
1756
+ chunksize: int = _DEFAULT_CHUNKSIZE,
1455
1757
  ) -> str:
1456
1758
  """Fit a scaler on an Ersilia output file and save the parameters.
1457
1759
 
1760
+ **Bounded memory.** The full matrix is never resident. The fit reads a
1761
+ batch of feature columns at a time (``_FIT_COLUMN_BATCH``) and fits each
1762
+ column from that in-memory block, so peak memory is the batch
1763
+ (~``batch × n_rows`` floats) — a few hundred MB — regardless of how wide
1764
+ the file is. H5 inputs are sliced directly; CSV inputs are first
1765
+ streamed — in
1766
+ row-chunks of *chunksize* — into a temporary columnar H5 (``float32``),
1767
+ which is then column-sliced for the fit and removed afterwards; this is
1768
+ the one unavoidable full read of the CSV, done without loading it whole.
1769
+
1458
1770
  The scaler JSON written to *scaler_path* is dtype-agnostic — see
1459
1771
  :func:`fit` for the parameter set. When *output_path* is provided
1460
1772
  the scaled data is also written immediately (fit-then-transform in
1461
- one call), and *output_dtype* selects the dtype of that inline
1462
- output. The dtype is **not** recorded in the scaler JSON; later
1463
- calls to :func:`transform_file` choose the dtype independently.
1773
+ one call, streamed row-chunk by row-chunk), and *output_dtype* selects
1774
+ the dtype of that inline output. The dtype is **not** recorded in the
1775
+ scaler JSON; later calls to :func:`transform_file` choose the dtype
1776
+ independently.
1464
1777
 
1465
1778
  The scaler filename's encoded model ID and version must match the
1466
1779
  input file's. The transformer JSON records ``eosframes_version``
@@ -1487,6 +1800,10 @@ def fit_file(
1487
1800
  Only used for the inline transform when *output_path* is
1488
1801
  given. ``"int8"`` quantizes the scaled values into
1489
1802
  ``[-127, 127]`` with sentinel ``-128`` for missing.
1803
+ chunksize : int, default ``50_000``
1804
+ Rows per chunk for the CSV→H5 staging pass and the inline
1805
+ transform. Bounds peak memory; tune down for extremely wide
1806
+ frames, up for narrow ones.
1490
1807
 
1491
1808
  Returns
1492
1809
  -------
@@ -1540,30 +1857,71 @@ def fit_file(
1540
1857
  f"Output file '{output_path}' already exists. Remove it first."
1541
1858
  )
1542
1859
 
1543
- from .ops import _read_file
1544
-
1545
- df = _read_file(input_path)
1546
-
1547
1860
  if output_dtype not in _VALID_OUTPUT_DTYPES:
1548
1861
  raise EosframesError(
1549
- f"Unknown output_dtype '{output_dtype}'. "
1550
- f"Supported: {_VALID_OUTPUT_DTYPES}"
1862
+ f"Unknown output_dtype '{output_dtype}'. Supported: {_VALID_OUTPUT_DTYPES}"
1551
1863
  )
1552
1864
 
1865
+ in_ext = os.path.splitext(input_path)[1].lower()
1553
1866
  mode_label = "fit + transform" if output_path is not None else "fit only"
1554
- logger.info(
1555
- "%s — reading %d rows from %s", mode_label, len(df), input_path
1556
- )
1557
- fitted = fit(df)
1867
+
1868
+ # Resolve a column-sliceable H5 to fit from. H5 inputs are used as-is;
1869
+ # CSV inputs are streamed into a temporary columnar H5 first.
1870
+ staged_tmp: Optional[str] = None
1871
+ try:
1872
+ if in_ext == ".h5":
1873
+ fit_source = input_path
1874
+ feature_cols = _h5_feature_names(input_path)
1875
+ n_rows = _h5_nrows(input_path)
1876
+ elif in_ext == ".csv":
1877
+ fd, staged_tmp = tempfile.mkstemp(
1878
+ suffix=".h5", dir=os.path.dirname(os.path.abspath(input_path))
1879
+ )
1880
+ os.close(fd)
1881
+ logger.info(
1882
+ "%s — staging CSV → temporary columnar H5 for column-wise fit "
1883
+ "(chunksize=%d)",
1884
+ mode_label,
1885
+ chunksize,
1886
+ )
1887
+ feature_cols, n_rows = _stage_csv_to_h5(input_path, staged_tmp, chunksize)
1888
+ fit_source = staged_tmp
1889
+ else:
1890
+ raise EosframesError(
1891
+ f"Unsupported input format '{in_ext}'. Expected .csv or .h5"
1892
+ )
1893
+
1894
+ if not feature_cols:
1895
+ raise EosframesError("No numeric feature columns found to fit the scaler.")
1896
+
1897
+ logger.info(
1898
+ "%s — fitting %d columns over %s rows (batches of %d)",
1899
+ mode_label,
1900
+ len(feature_cols),
1901
+ f"{n_rows:,}",
1902
+ _FIT_COLUMN_BATCH,
1903
+ )
1904
+ columns: dict = {}
1905
+ for name, arr in progress(
1906
+ _iter_h5_columns(fit_source),
1907
+ total=len(feature_cols),
1908
+ desc="Fitting columns",
1909
+ log_every=max(1, len(feature_cols) // _PROGRESS_LOG_STEPS),
1910
+ ):
1911
+ columns[name] = _fit_series(pd.Series(arr))
1912
+ _log_fit_summary(columns, logger)
1913
+ finally:
1914
+ if staged_tmp is not None and os.path.exists(staged_tmp):
1915
+ os.remove(staged_tmp)
1558
1916
 
1559
1917
  transformer = {
1560
1918
  "eosframes_version": _PACKAGE_VERSION,
1561
- "method": fitted["method"],
1919
+ "method": _METHOD_NAME,
1562
1920
  "model_id": parsed["model_id"],
1563
1921
  "model_version": parsed["version"],
1564
1922
  "fitted_at": datetime.now().isoformat(timespec="seconds"),
1565
- "n_rows": len(df),
1566
- "columns": fitted["columns"],
1923
+ "n_rows": n_rows,
1924
+ "columns": columns,
1567
1925
  }
1568
1926
  with open(scaler_path, "w") as fh:
1569
1927
  json.dump(transformer, fh, indent=2)
@@ -1575,15 +1933,15 @@ def fit_file(
1575
1933
  output_dtype,
1576
1934
  impute,
1577
1935
  )
1578
- scaled_df = transform(
1579
- df, transformer, output_dtype=output_dtype, impute=impute
1580
- )
1581
- _write_df(
1582
- scaled_df,
1936
+ written = _streaming_transform(
1937
+ input_path,
1938
+ transformer,
1583
1939
  output_path,
1584
- values_dtype=_values_dtype_for(output_dtype),
1940
+ output_dtype,
1941
+ impute,
1942
+ chunksize,
1585
1943
  )
1586
- logger.info("Scaled output written to %s", output_path)
1944
+ logger.info("Scaled output written to %s (%d rows)", output_path, written)
1587
1945
 
1588
1946
  return scaler_path
1589
1947
 
@@ -1594,9 +1952,15 @@ def transform_file(
1594
1952
  output_path: str,
1595
1953
  output_dtype: str = _DEFAULT_OUTPUT_DTYPE,
1596
1954
  impute: bool = False,
1955
+ chunksize: int = _DEFAULT_CHUNKSIZE,
1597
1956
  ) -> str:
1598
1957
  """Apply a saved scaler to an Ersilia output file.
1599
1958
 
1959
+ **Bounded memory.** The input is read and the output written one
1960
+ row-chunk of *chunksize* at a time, so peak memory is a single chunk
1961
+ in plus a single chunk out — independent of the file's size. Works for
1962
+ CSV and H5 inputs and outputs in any combination.
1963
+
1600
1964
  The scaler's recorded ``eosframes_version`` must exactly match the
1601
1965
  running ``eosframes.__version__``, and ``method`` must still be
1602
1966
  ``"robust_typed"``; any mismatch raises with a clear "re-fit"
@@ -1618,6 +1982,9 @@ def transform_file(
1618
1982
  output_dtype : {"float32", "int8"}, default ``"float32"``
1619
1983
  ``"int8"`` quantizes scaled values to ``[-127, 127]`` with
1620
1984
  sentinel ``-128`` for missing.
1985
+ chunksize : int, default ``50_000``
1986
+ Rows per streamed chunk. Bounds peak memory; tune down for
1987
+ extremely wide frames, up for narrow ones.
1621
1988
 
1622
1989
  Returns
1623
1990
  -------
@@ -1673,8 +2040,7 @@ def transform_file(
1673
2040
 
1674
2041
  if output_dtype not in _VALID_OUTPUT_DTYPES:
1675
2042
  raise EosframesError(
1676
- f"Unknown output_dtype '{output_dtype}'. "
1677
- f"Supported: {_VALID_OUTPUT_DTYPES}"
2043
+ f"Unknown output_dtype '{output_dtype}'. Supported: {_VALID_OUTPUT_DTYPES}"
1678
2044
  )
1679
2045
 
1680
2046
  t_model_id = transformer.get("model_id")
@@ -1693,21 +2059,21 @@ def transform_file(
1693
2059
  f"scaler was fitted on '{t_model_version}'."
1694
2060
  )
1695
2061
 
1696
- from .ops import _read_file
1697
-
1698
- df = _read_file(input_path)
1699
-
1700
2062
  logger.info(
1701
- "Applying scaler to %d rows from %s (output_dtype=%s, impute=%s)",
1702
- len(df),
2063
+ "Applying scaler to %s in row-chunks (chunksize=%d, output_dtype=%s, "
2064
+ "impute=%s)",
1703
2065
  input_path,
2066
+ chunksize,
1704
2067
  output_dtype,
1705
2068
  impute,
1706
2069
  )
1707
- scaled_df = transform(
1708
- df, transformer, output_dtype=output_dtype, impute=impute
2070
+ written = _streaming_transform(
2071
+ input_path,
2072
+ transformer,
2073
+ output_path,
2074
+ output_dtype,
2075
+ impute,
2076
+ chunksize,
1709
2077
  )
1710
-
1711
- _write_df(scaled_df, output_path, values_dtype=_values_dtype_for(output_dtype))
1712
- logger.info("Scaled output written to %s", output_path)
2078
+ logger.info("Scaled output written to %s (%d rows)", output_path, written)
1713
2079
  return output_path
@@ -0,0 +1,126 @@
1
+ """General-purpose utilities."""
2
+
3
+ import logging
4
+ import sys
5
+ import time
6
+ from typing import Iterable, Iterator, Optional, TypeVar
7
+
8
+ import pandas as pd
9
+
10
+ from .logger import get_logger
11
+
12
+ _T = TypeVar("_T")
13
+
14
+
15
+ def chunker(df: pd.DataFrame, chunksize: int = 10000) -> Iterator[pd.DataFrame]:
16
+ """Yield successive non-overlapping chunks of *df*.
17
+
18
+ Parameters
19
+ ----------
20
+ df : pd.DataFrame
21
+ The DataFrame to split.
22
+ chunksize : int
23
+ Number of rows per chunk (default 10 000).
24
+
25
+ Yields
26
+ ------
27
+ pd.DataFrame
28
+ """
29
+ for start in range(0, len(df), chunksize):
30
+ yield df.iloc[start : start + chunksize]
31
+
32
+
33
+ def progress(
34
+ iterable: Iterable[_T],
35
+ total: Optional[int] = None,
36
+ desc: str = "",
37
+ log_every: Optional[int] = None,
38
+ ) -> Iterator[_T]:
39
+ """Wrap *iterable* with progress feedback that adapts to the environment.
40
+
41
+ Three modes, picked automatically (all gated on the ``eosframes`` logger
42
+ being at ``INFO`` or more verbose — a quieted logger gets a silent
43
+ pass-through):
44
+
45
+ * **Interactive (TTY)** — an in-place per-item bar on stderr.
46
+ * **Non-interactive with** *log_every* — periodic ``INFO`` log lines
47
+ (every *log_every* items, plus a final line). This is the path that
48
+ keeps long streaming jobs legible when stderr is a file or pipe:
49
+ ``nohup``, CI, ``build_scaler.sh``, etc. Without *log_every* a
50
+ non-interactive run stays silent.
51
+ * **Otherwise** — transparent pass-through, no output.
52
+
53
+ Both visible modes share stderr with the logger, so nothing ever leaks
54
+ onto piped stdout.
55
+
56
+ Parameters
57
+ ----------
58
+ iterable : Iterable
59
+ The items to iterate over.
60
+ total : int, optional
61
+ Expected number of items. Inferred via ``len(iterable)`` when omitted;
62
+ if neither is available the bar is disabled (percentages are dropped
63
+ from log lines but counts still flow).
64
+ desc : str
65
+ Short label shown to the left of the bar / log line.
66
+ log_every : int, optional
67
+ Emit an ``INFO`` log line every this many items when no bar is drawn.
68
+ Leave unset for the in-memory paths that should stay quiet off-TTY.
69
+
70
+ Yields
71
+ ------
72
+ The items of *iterable*, unchanged.
73
+ """
74
+ if total is None:
75
+ try:
76
+ total = len(iterable) # type: ignore[arg-type]
77
+ except TypeError:
78
+ total = None
79
+
80
+ logger = get_logger()
81
+ info_on = logger.isEnabledFor(logging.INFO)
82
+ bar_on = total is not None and total > 1 and sys.stderr.isatty() and info_on
83
+
84
+ if bar_on:
85
+ width = 30
86
+ last_draw = -1.0 # force an immediate first draw
87
+ count = 0
88
+ for count, item in enumerate(iterable, 1):
89
+ yield item
90
+ now = time.monotonic()
91
+ # Throttle redraws to ~10 fps, but always paint the final item.
92
+ if count < total and now - last_draw < 0.1:
93
+ continue
94
+ last_draw = now
95
+ frac = count / total
96
+ filled = int(width * frac)
97
+ bar = "█" * filled + "·" * (width - filled)
98
+ label = f"{desc} " if desc else ""
99
+ sys.stderr.write(f"\r{label}|{bar}| {count}/{total}")
100
+ sys.stderr.flush()
101
+ if count:
102
+ sys.stderr.write("\n")
103
+ sys.stderr.flush()
104
+ return
105
+
106
+ if log_every and info_on:
107
+ label = desc or "progress"
108
+ count = 0
109
+ for count, item in enumerate(iterable, 1):
110
+ yield item
111
+ if count % log_every == 0:
112
+ if total:
113
+ logger.info(
114
+ "%s: %d/%d (%d%%)", label, count, total, count * 100 // total
115
+ )
116
+ else:
117
+ logger.info("%s: %d done", label, count)
118
+ # Final line unless the last item already landed on a logged boundary.
119
+ if count and count % log_every != 0:
120
+ if total:
121
+ logger.info("%s: %d/%d (100%%)", label, count, total)
122
+ else:
123
+ logger.info("%s: %d done", label, count)
124
+ return
125
+
126
+ yield from iterable
@@ -1,23 +0,0 @@
1
- """General-purpose utilities."""
2
-
3
- from typing import Iterator
4
-
5
- import pandas as pd
6
-
7
-
8
- def chunker(df: pd.DataFrame, chunksize: int = 10000) -> Iterator[pd.DataFrame]:
9
- """Yield successive non-overlapping chunks of *df*.
10
-
11
- Parameters
12
- ----------
13
- df : pd.DataFrame
14
- The DataFrame to split.
15
- chunksize : int
16
- Number of rows per chunk (default 10 000).
17
-
18
- Yields
19
- ------
20
- pd.DataFrame
21
- """
22
- for start in range(0, len(df), chunksize):
23
- yield df.iloc[start : start + chunksize]
File without changes