eolas-data 1.3.4__tar.gz → 1.3.13__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {eolas_data-1.3.4 → eolas_data-1.3.13}/.github/workflows/smoke.yml +2 -2
  2. {eolas_data-1.3.4 → eolas_data-1.3.13}/PKG-INFO +2 -2
  3. {eolas_data-1.3.4 → eolas_data-1.3.13}/README.md +1 -1
  4. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/__init__.py +1 -1
  5. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/client.py +168 -57
  6. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/library.py +93 -0
  7. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/meta.py +59 -5
  8. {eolas_data-1.3.4 → eolas_data-1.3.13}/pyproject.toml +1 -1
  9. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_meta.py +26 -0
  10. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_progress.py +57 -0
  11. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_smoke_live.py +32 -3
  12. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_sync_bulk.py +59 -0
  13. {eolas_data-1.3.4 → eolas_data-1.3.13}/.github/workflows/catalog-drift.yml +0 -0
  14. {eolas_data-1.3.4 → eolas_data-1.3.13}/.github/workflows/publish.yml +0 -0
  15. {eolas_data-1.3.4 → eolas_data-1.3.13}/.github/workflows/test.yml +0 -0
  16. {eolas_data-1.3.4 → eolas_data-1.3.13}/.gitignore +0 -0
  17. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/_dataset_names.py +0 -0
  18. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/_regen_names.py +0 -0
  19. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/cdc.py +0 -0
  20. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/cli.py +0 -0
  21. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/console.py +0 -0
  22. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/dataset.py +0 -0
  23. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/exceptions.py +0 -0
  24. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/rows.py +0 -0
  25. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/schedule.py +0 -0
  26. {eolas_data-1.3.4 → eolas_data-1.3.13}/eolas_data/search.py +0 -0
  27. {eolas_data-1.3.4 → eolas_data-1.3.13}/scripts/preflight.sh +0 -0
  28. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_as_arrow.py +0 -0
  29. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_cdc_roundtrip.py +0 -0
  30. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_cli.py +0 -0
  31. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_client.py +0 -0
  32. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_keyring.py +0 -0
  33. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_library.py +0 -0
  34. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_rows.py +0 -0
  35. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_schedule.py +0 -0
  36. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_search.py +0 -0
  37. {eolas_data-1.3.4 → eolas_data-1.3.13}/tests/test_sync_changes.py +0 -0
@@ -15,8 +15,8 @@ jobs:
15
15
  with:
16
16
  python-version: "3.12"
17
17
 
18
- - name: Install package + dev deps
19
- run: pip install -e ".[dev]"
18
+ - name: Install package + dev/geo deps (as users would)
19
+ run: pip install ".[dev,geo]"
20
20
 
21
21
  - name: Live smoke tests
22
22
  env:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: eolas-data
3
- Version: 1.3.4
3
+ Version: 1.3.13
4
4
  Summary: Python client for the eolas.fyi statistical data API (NZ, Australia, OECD)
5
5
  Project-URL: Homepage, https://eolas.fyi
6
6
  Project-URL: Documentation, https://docs.eolas.fyi/
@@ -231,7 +231,7 @@ r = client.sync_bulk("nz_cpi", path="nz_cpi.parquet")
231
231
  path = client.download_bulk("treasury_fiscal_spending", path="t.parquet")
232
232
  ```
233
233
 
234
- **Progress bars:** `download_bulk`, `sync_bulk`, and `get_local` all show a `tqdm` progress bar automatically in interactive terminals and VSCode notebooks, so 1+ GB files are never silent. Pass `progress=False` to suppress in scripts, or set `EOLAS_NO_PROGRESS=1` in the environment for a CI-wide escape hatch. The `--no-progress` flag does the same from the CLI.
234
+ **Progress bars:** `get_local()` shows two phases in interactive sessions — a **download** byte bar while fetching from CDN, then a **read** spinner while Parquet/GeoParquet is loaded (often the slow part on multi-million-row geo datasets). Control with `progress=True` (both), `False` (neither), `"download"`, or `"read"`. Set `EOLAS_NO_PROGRESS=1` to suppress both in batch scripts. Cached files skip the download bar and print an informative message instead.
235
235
 
236
236
  CLI mirror: `eolas download <name>` for one-shot, `eolas sync <name> [--watch hourly]` for an incremental check. Full docs: [docs.eolas.fyi/bulk-downloads/](https://docs.eolas.fyi/bulk-downloads/).
237
237
 
@@ -183,7 +183,7 @@ r = client.sync_bulk("nz_cpi", path="nz_cpi.parquet")
183
183
  path = client.download_bulk("treasury_fiscal_spending", path="t.parquet")
184
184
  ```
185
185
 
186
- **Progress bars:** `download_bulk`, `sync_bulk`, and `get_local` all show a `tqdm` progress bar automatically in interactive terminals and VSCode notebooks, so 1+ GB files are never silent. Pass `progress=False` to suppress in scripts, or set `EOLAS_NO_PROGRESS=1` in the environment for a CI-wide escape hatch. The `--no-progress` flag does the same from the CLI.
186
+ **Progress bars:** `get_local()` shows two phases in interactive sessions — a **download** byte bar while fetching from CDN, then a **read** spinner while Parquet/GeoParquet is loaded (often the slow part on multi-million-row geo datasets). Control with `progress=True` (both), `False` (neither), `"download"`, or `"read"`. Set `EOLAS_NO_PROGRESS=1` to suppress both in batch scripts. Cached files skip the download bar and print an informative message instead.
187
187
 
188
188
  CLI mirror: `eolas download <name>` for one-shot, `eolas sync <name> [--watch hourly]` for an incremental check. Full docs: [docs.eolas.fyi/bulk-downloads/](https://docs.eolas.fyi/bulk-downloads/).
189
189
 
@@ -15,7 +15,7 @@ from .exceptions import (
15
15
  WatermarkExpired,
16
16
  )
17
17
 
18
- __version__ = "1.3.3"
18
+ __version__ = "1.3.13"
19
19
 
20
20
  __all__ = [
21
21
  "Client",
@@ -1,5 +1,6 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import contextlib
3
4
  import datetime
4
5
  import json
5
6
  import logging
@@ -7,7 +8,10 @@ import os
7
8
  import pathlib
8
9
  import sys
9
10
  from dataclasses import dataclass
10
- from typing import Optional, Union
11
+ from typing import Literal, Optional, Union
12
+
13
+ ProgressControl = Union[bool, str, None]
14
+ ProgressPhase = Literal["download", "read"]
11
15
 
12
16
  import pandas as pd
13
17
  import requests
@@ -262,6 +266,36 @@ class Client:
262
266
  self._meta_cache[key] = self.info(name)
263
267
  return self._meta_cache[key]
264
268
 
269
+ def _apply_force(self, name: Union[str, "DatasetName"], force: bool) -> None:
270
+ if force:
271
+ self._meta_cache.pop(str(name), None)
272
+
273
+ def cache_clear(
274
+ self,
275
+ name: Optional[Union[str, "DatasetName"]] = None,
276
+ *,
277
+ cache_dir: Optional[Union[str, "pathlib.Path"]] = None,
278
+ format: Optional[str] = None,
279
+ files: bool = True,
280
+ meta: bool = True,
281
+ ) -> dict:
282
+ """Clear client-side cache (library files and/or session metadata).
283
+
284
+ See :func:`eolas_data.library.cache_clear` for details. Pass
285
+ ``force=True`` to :meth:`get_local` / :meth:`sync_bulk` to clear
286
+ metadata and re-download in one step.
287
+ """
288
+ from .library import cache_clear as _cache_clear
289
+
290
+ return _cache_clear(
291
+ None if name is None else str(name),
292
+ cache_dir=cache_dir,
293
+ format=format,
294
+ files=files,
295
+ meta=meta,
296
+ meta_cache=self._meta_cache if meta else None,
297
+ )
298
+
265
299
  def _attach_dataset_meta(
266
300
  self,
267
301
  result: "pd.DataFrame",
@@ -362,7 +396,7 @@ class Client:
362
396
  freshness: str = "auto",
363
397
  format: str = "parquet",
364
398
  path: Optional[Union[str, "pathlib.Path"]] = None,
365
- progress: Optional[bool] = None,
399
+ progress: ProgressControl = None,
366
400
  ) -> "Union[pathlib.Path, bytes]":
367
401
  """Download a complete dataset as a single binary file.
368
402
 
@@ -387,14 +421,10 @@ class Client:
387
421
  bytes. Pass a ``str`` or ``pathlib.Path`` to write to disk and
388
422
  return the resolved path. Parent directories are created if
389
423
  needed.
390
- progress: Control the download progress bar.
391
- ``None`` (default) auto-detects: bar shown when
392
- ``sys.stdout.isatty()`` is True (interactive terminal or
393
- notebook), hidden when piped or in CI.
394
- ``True`` forces the bar on regardless (useful in log-tailing
395
- scenarios). ``False`` forces it off. When ``path`` is
396
- ``None`` (bytes mode) progress is always disabled — there is
397
- no file path to label.
424
+ progress: Control the download progress bar (``"download"`` phase).
425
+ See :meth:`get_local` for the full selector vocabulary
426
+ (``"read"`` applies only to :meth:`get_local`). When ``path``
427
+ is ``None`` (bytes mode) progress is always disabled.
398
428
 
399
429
  Returns:
400
430
  ``pathlib.Path`` when ``path`` is set (the resolved, written path).
@@ -467,13 +497,13 @@ class Client:
467
497
  out = pathlib.Path(path).expanduser().resolve()
468
498
  out.parent.mkdir(parents=True, exist_ok=True)
469
499
 
470
- show = self._resolve_show_progress(progress)
500
+ show = self._resolve_show_progress(progress, "download")
471
501
  resp = self._raw_bulk_get(bulk_path, params=params, stream=True)
472
502
  total = int(resp.headers.get("Content-Length", 0)) or None
473
503
  self._stream_to_file_with_progress(
474
504
  resp, out,
475
505
  total_bytes=total,
476
- label=out.name,
506
+ label=f"Downloading {out.name}",
477
507
  show_progress=show,
478
508
  )
479
509
  return out
@@ -485,7 +515,8 @@ class Client:
485
515
  path: Union[str, "pathlib.Path"],
486
516
  format: str = "parquet",
487
517
  freshness: str = "auto",
488
- progress: Optional[bool] = None,
518
+ progress: ProgressControl = None,
519
+ force: bool = False,
489
520
  ) -> SyncResult:
490
521
  """Incrementally sync a bulk dataset file — only re-download when the snapshot changes.
491
522
 
@@ -514,11 +545,12 @@ class Client:
514
545
  ``"geoparquet"``.
515
546
  freshness: ``"auto"`` (default), ``"monthly"``, or ``"current"``.
516
547
  Passed verbatim to the bulk endpoint.
517
- progress: Control the download progress bar.
518
- ``None`` (default) auto-detects via ``sys.stdout.isatty()``.
519
- ``True`` forces the bar on; ``False`` forces it off.
520
- When ``status="unchanged"`` no data is transferred so no bar
521
- is shown regardless of this setting.
548
+ progress: Control the download progress bar (``"download"`` phase).
549
+ See :meth:`get_local` for the full selector vocabulary.
550
+ When ``status="unchanged"`` no download bar is shown; an
551
+ informative cached-file message is printed instead.
552
+ force: When ``True``, skip the sidecar unchanged fast path and
553
+ re-download even when the local snapshot id matches the server.
522
554
 
523
555
  Returns:
524
556
  A :class:`SyncResult` dataclass with ``status``,
@@ -566,6 +598,8 @@ class Client:
566
598
  f"Unknown freshness {freshness!r}. Expected 'auto', 'monthly', or 'current'."
567
599
  )
568
600
 
601
+ self._apply_force(name, force)
602
+
569
603
  out = pathlib.Path(path).expanduser().resolve()
570
604
  sidecar = pathlib.Path(str(out) + ".eolas-meta.json")
571
605
 
@@ -602,10 +636,12 @@ class Client:
602
636
 
603
637
  # No-op fast path: snapshot hasn't changed AND file exists on disk.
604
638
  if (
605
- prev is not None
639
+ not force
640
+ and prev is not None
606
641
  and prev.get("snapshot_id") == current_sid
607
642
  and out.exists()
608
643
  ):
644
+ print(f"Using cached {out.name} (up to date).", file=sys.stderr)
609
645
  return SyncResult(
610
646
  status="unchanged",
611
647
  previous_snapshot_id=prev.get("snapshot_id"),
@@ -617,14 +653,14 @@ class Client:
617
653
  # Download (atomic replace).
618
654
  out.parent.mkdir(parents=True, exist_ok=True)
619
655
  tmp = out.with_suffix(out.suffix + f".eolas-tmp-{os.urandom(4).hex()}")
620
- show = self._resolve_show_progress(progress)
656
+ show = self._resolve_show_progress(progress, "download")
621
657
  try:
622
658
  resp = self._raw_bulk_get(bulk_path, params=params, stream=True)
623
659
  total = int(resp.headers.get("Content-Length", 0)) or None
624
660
  bytes_dl = self._stream_to_file_with_progress(
625
661
  resp, tmp,
626
662
  total_bytes=total,
627
- label=out.name,
663
+ label=f"Downloading {out.name}",
628
664
  show_progress=show,
629
665
  )
630
666
  if bytes_dl == 0:
@@ -677,6 +713,7 @@ class Client:
677
713
  *,
678
714
  format: str = "parquet",
679
715
  progress: Optional[bool] = None,
716
+ force: bool = False,
680
717
  ) -> SyncResult:
681
718
  """Incrementally sync a changelog-tier dataset via the /changes feed.
682
719
 
@@ -710,6 +747,7 @@ class Client:
710
747
  changelog sync; other formats raise ``ValueError``.
711
748
  progress: Forwarded to :meth:`sync_bulk` for the baseline download
712
749
  progress bar (``None`` auto-detects TTY).
750
+ force: When ``True``, re-baseline from a full bulk snapshot.
713
751
 
714
752
  Returns:
715
753
  A :class:`SyncResult` with ``sync_mode='changelog'``,
@@ -748,6 +786,7 @@ class Client:
748
786
 
749
787
  out = pathlib.Path(path).expanduser().resolve()
750
788
  sidecar_path = pathlib.Path(str(out) + ".eolas-meta.json")
789
+ self._apply_force(name, force)
751
790
 
752
791
  # Read sidecar.
753
792
  sidecar: Optional[dict] = None
@@ -758,7 +797,7 @@ class Client:
758
797
  sidecar = None
759
798
 
760
799
  # Determine if we need a cold-start baseline.
761
- needs_baseline = (
800
+ needs_baseline = force or (
762
801
  sidecar is None
763
802
  or sidecar.get("sync_mode") != "changelog"
764
803
  or sidecar.get("watermark_seq") is None
@@ -782,6 +821,7 @@ class Client:
782
821
  format=fmt,
783
822
  freshness="current",
784
823
  progress=progress,
824
+ force=force,
785
825
  )
786
826
  baseline_snapshot_id = bulk_result.current_snapshot_id
787
827
 
@@ -844,6 +884,7 @@ class Client:
844
884
  format=fmt,
845
885
  freshness="current",
846
886
  progress=progress,
887
+ force=force,
847
888
  )
848
889
  baseline_snapshot_id = bulk_result.current_snapshot_id
849
890
  high_seq = self._fetch_changes_seq_high(changes_url, since_seq=2**62)
@@ -948,6 +989,7 @@ class Client:
948
989
  format: str = "parquet",
949
990
  freshness: str = "auto",
950
991
  progress: Optional[bool] = None,
992
+ force: bool = False,
951
993
  ) -> SyncResult:
952
994
  """Unified sync dispatcher — keeps a local file current automatically.
953
995
 
@@ -973,6 +1015,7 @@ class Client:
973
1015
  freshness: ``"auto"`` (default), ``"monthly"``, or ``"current"``. Only
974
1016
  used for the snapshot path (passed to :meth:`sync_bulk`).
975
1017
  progress: Control the download progress bar.
1018
+ force: Bypass local unchanged cache and re-sync from the server.
976
1019
 
977
1020
  Returns:
978
1021
  A :class:`SyncResult`. The ``sync_mode`` field is ``"snapshot"`` or
@@ -996,7 +1039,7 @@ class Client:
996
1039
  tier = meta.get("cdc_serving_tier", "snapshot") or "snapshot"
997
1040
 
998
1041
  if tier == "changelog":
999
- return self.sync_changes(name, path, format=format, progress=progress)
1042
+ return self.sync_changes(name, path, format=format, progress=progress, force=force)
1000
1043
  else:
1001
1044
  result = self.sync_bulk(
1002
1045
  name,
@@ -1004,6 +1047,7 @@ class Client:
1004
1047
  format=format,
1005
1048
  freshness=freshness,
1006
1049
  progress=progress,
1050
+ force=force,
1007
1051
  )
1008
1052
  # Tag the result with sync_mode so callers don't need to inspect tier.
1009
1053
  result.sync_mode = "snapshot"
@@ -1159,28 +1203,73 @@ class Client:
1159
1203
  # ------------------------------------------------------------------
1160
1204
 
1161
1205
  @staticmethod
1162
- def _resolve_show_progress(progress: Optional[bool]) -> bool:
1163
- """Resolve the tri-state ``progress`` kwarg to a concrete bool.
1164
-
1165
- Priority (highest first):
1166
- 1. Explicit ``progress=True/False`` kwarg from the caller.
1167
- 2. ``EOLAS_NO_PROGRESS=1`` environment variable — always suppresses.
1168
- 3. Jupyter / IPython kernel detection — Jupyter wraps stdout in an
1169
- ``OutStream``, so ``sys.stdout.isatty()`` returns False even when
1170
- the user is clearly in an interactive notebook session. If
1171
- ``ipykernel`` is loaded (the standard signal used by tqdm itself),
1172
- we're in a notebook → show the bar; ``tqdm.auto`` then renders it
1173
- as an ipywidget (or text bar as fallback).
1174
- 4. ``sys.stdout.isatty()`` auto-detection for plain terminals.
1175
- """
1176
- if progress is not None:
1177
- return bool(progress)
1206
+ def _progress_auto_detect() -> bool:
1207
+ """Whether progress feedback should show when ``progress`` is None."""
1178
1208
  if os.getenv("EOLAS_NO_PROGRESS", "").strip() in ("1", "true", "yes"):
1179
1209
  return False
1180
1210
  if "ipykernel" in sys.modules:
1181
1211
  return True
1182
1212
  return sys.stdout.isatty()
1183
1213
 
1214
+ @staticmethod
1215
+ def _resolve_progress_phases(progress: ProgressControl) -> dict[str, bool]:
1216
+ """Map ``progress`` to download/read phase flags.
1217
+
1218
+ Accepts ``None``, ``True``/``False``, or ``"both"``, ``"download"``,
1219
+ ``"read"``, ``"none"`` (and ``"all"`` as alias for ``"both"``).
1220
+ """
1221
+ if progress is False:
1222
+ return {"download": False, "read": False}
1223
+ if progress is True:
1224
+ return {"download": True, "read": True}
1225
+ if isinstance(progress, str):
1226
+ key = progress.strip().lower()
1227
+ table = {
1228
+ "both": (True, True),
1229
+ "all": (True, True),
1230
+ "download": (True, False),
1231
+ "read": (False, True),
1232
+ "none": (False, False),
1233
+ }
1234
+ if key not in table:
1235
+ raise ValueError(
1236
+ "progress must be True, False, None, or one of "
1237
+ "'both', 'download', 'read', 'none'."
1238
+ )
1239
+ d, r = table[key]
1240
+ return {"download": d, "read": r}
1241
+ auto = Client._progress_auto_detect()
1242
+ return {"download": auto, "read": auto}
1243
+
1244
+ @staticmethod
1245
+ def _resolve_show_progress(
1246
+ progress: ProgressControl,
1247
+ phase: ProgressPhase = "download",
1248
+ ) -> bool:
1249
+ """Resolve ``progress`` for one bulk phase (download or read)."""
1250
+ return Client._resolve_progress_phases(progress)[phase]
1251
+
1252
+ @staticmethod
1253
+ @contextlib.contextmanager
1254
+ def _with_read_progress(label: str, show: bool):
1255
+ """Indeterminate spinner while materialising a cached bulk file."""
1256
+ if not show:
1257
+ yield
1258
+ return
1259
+ import tqdm.auto
1260
+
1261
+ bar = tqdm.auto.tqdm(
1262
+ total=1,
1263
+ desc=f"Loading {label} from disk",
1264
+ leave=False,
1265
+ bar_format="{desc}…",
1266
+ )
1267
+ try:
1268
+ yield
1269
+ finally:
1270
+ bar.update(1)
1271
+ bar.close()
1272
+
1184
1273
  @staticmethod
1185
1274
  def _stream_to_file_with_progress(
1186
1275
  resp: "requests.Response",
@@ -1796,7 +1885,8 @@ class Client:
1796
1885
  as_geo: Optional[bool] = None,
1797
1886
  as_arrow: bool = False,
1798
1887
  meta: bool = True,
1799
- progress: Optional[bool] = None,
1888
+ progress: ProgressControl = None,
1889
+ force: bool = False,
1800
1890
  ) -> "pd.DataFrame":
1801
1891
  """Download (or serve from cache) a whole dataset as a local DataFrame.
1802
1892
 
@@ -1833,7 +1923,16 @@ class Client:
1833
1923
  meta: When ``True`` (default), attach table/column metadata from
1834
1924
  :meth:`info` (session-cached). Pass ``False`` to skip the extra
1835
1925
  round-trip.
1836
- progress: Forwarded to :meth:`sync_bulk` (``None`` auto-detects TTY).
1926
+ progress: Control progress for two phases: **download** (streaming
1927
+ byte bar via :meth:`sync_bulk`) and **read** (spinner while
1928
+ Parquet/GeoParquet is loaded into a DataFrame). ``None``
1929
+ (default) enables both in interactive sessions. ``True``/``False``
1930
+ force both on/off. Use ``"download"``, ``"read"``, ``"both"``,
1931
+ or ``"none"`` for one phase only. Suppressed by
1932
+ ``EOLAS_NO_PROGRESS=1``. Cached snapshots skip the download bar
1933
+ and print an informative message instead.
1934
+ force: Re-download the library file even when the sidecar says the
1935
+ snapshot is current. See :meth:`cache_clear`.
1837
1936
 
1838
1937
  Returns:
1839
1938
  ``pd.DataFrame`` (tabular) or ``geopandas.GeoDataFrame`` (geo +
@@ -1856,6 +1955,7 @@ class Client:
1856
1955
  )
1857
1956
  # Resolve as_geo: None → True (auto) unless as_arrow overrides.
1858
1957
  resolved_as_geo = as_geo if as_geo is not None else (not as_arrow)
1958
+ self._apply_force(name, force)
1859
1959
 
1860
1960
  # ---- resolve cache_dir -----------------------------------------------
1861
1961
  if cache_dir is not None:
@@ -1894,26 +1994,36 @@ class Client:
1894
1994
  file_path = cache_path / f"{name}{ext}"
1895
1995
 
1896
1996
  # ---- sync (download if needed, HEAD check if cached) -----------------
1897
- self.sync_bulk(name, path=file_path, format=fmt, freshness=freshness, progress=progress)
1997
+ self.sync_bulk(
1998
+ name, path=file_path, format=fmt, freshness=freshness,
1999
+ progress=progress, force=force,
2000
+ )
2001
+
2002
+ show_read = self._resolve_show_progress(progress, "read")
2003
+ read_label = file_path.name
1898
2004
 
1899
2005
  # ---- read the local file into a DataFrame ----------------------------
1900
2006
  if as_arrow:
1901
2007
  import pyarrow.parquet as _pq
1902
- if fmt == "csv_gz":
1903
- import pyarrow as _pa
1904
- return _pa.Table.from_pandas(pd.read_csv(file_path), preserve_index=False)
1905
- return _pq.read_table(file_path)
2008
+ with self._with_read_progress(read_label, show_read):
2009
+ if fmt == "csv_gz":
2010
+ import pyarrow as _pa
2011
+ return _pa.Table.from_pandas(
2012
+ pd.read_csv(file_path), preserve_index=False,
2013
+ )
2014
+ return _pq.read_table(file_path)
1906
2015
 
1907
2016
  def _read_bulk_file(path: pathlib.Path, bulk_fmt: str):
1908
- if bulk_fmt == "csv_gz":
1909
- return pd.read_csv(path)
1910
- if bulk_fmt == "geoparquet" and resolved_as_geo:
1911
- try:
1912
- import geopandas as gpd
1913
- return gpd.read_parquet(path)
1914
- except ImportError:
1915
- pass
1916
- return pd.read_parquet(path)
2017
+ with self._with_read_progress(read_label, show_read):
2018
+ if bulk_fmt == "csv_gz":
2019
+ return pd.read_csv(path)
2020
+ if bulk_fmt == "geoparquet" and resolved_as_geo:
2021
+ try:
2022
+ import geopandas as gpd
2023
+ return gpd.read_parquet(path)
2024
+ except ImportError:
2025
+ pass
2026
+ return pd.read_parquet(path)
1917
2027
 
1918
2028
  try:
1919
2029
  result = _read_bulk_file(file_path, fmt)
@@ -1929,7 +2039,7 @@ class Client:
1929
2039
  csv_path = cache_path / f"{name}{self._BULK_EXTENSIONS['csv_gz']}"
1930
2040
  self.sync_bulk(
1931
2041
  name, path=csv_path, format="csv_gz",
1932
- freshness=freshness, progress=progress,
2042
+ freshness=freshness, progress=progress, force=force,
1933
2043
  )
1934
2044
  result = _read_bulk_file(csv_path, "csv_gz")
1935
2045
  else:
@@ -1952,6 +2062,7 @@ class Client:
1952
2062
  as_arrow: bool = False,
1953
2063
  meta: bool = True,
1954
2064
  envelope: bool = False,
2065
+ force: bool = False,
1955
2066
  ) -> Dataset:
1956
2067
  """Fetch dataset rows as a pandas (or polars / geopandas) DataFrame.
1957
2068
 
@@ -2037,7 +2148,7 @@ class Client:
2037
2148
  info_meta = self._info_cached(name)
2038
2149
  if self._bulk_export_allowed(info_meta) and self._live_pull_blocked(info_meta):
2039
2150
  result = self.get_local(
2040
- name, as_geo=as_geo, as_arrow=False, meta=meta,
2151
+ name, as_geo=as_geo, as_arrow=False, meta=meta, force=force,
2041
2152
  )
2042
2153
  if engine == "polars":
2043
2154
  try:
@@ -39,6 +39,13 @@ _CONFIG_FILE = _CONFIG_DIR / "config.json"
39
39
  # Fallback cache directory (Step 5)
40
40
  _FALLBACK_DIR = pathlib.Path.home() / ".cache" / "eolas"
41
41
 
42
+ # Bulk file extensions (mirrors Client._BULK_EXTENSIONS).
43
+ _BULK_EXTENSIONS = {
44
+ "parquet": ".parquet",
45
+ "csv_gz": ".csv.gz",
46
+ "geoparquet": ".geo.parquet",
47
+ }
48
+
42
49
 
43
50
  # ---------------------------------------------------------------------------
44
51
  # Public API
@@ -94,6 +101,92 @@ def library_clear() -> None:
94
101
  _remove_config_key("library_dir")
95
102
 
96
103
 
104
+ def cache_clear(
105
+ name: Optional[str] = None,
106
+ *,
107
+ cache_dir: Optional[str | pathlib.Path] = None,
108
+ format: Optional[str] = None,
109
+ files: bool = True,
110
+ meta: bool = True,
111
+ meta_cache: Optional[dict] = None,
112
+ ) -> dict:
113
+ """Clear client-side cache for a dataset (or the whole library).
114
+
115
+ eolas-data caches at two client-only levels:
116
+
117
+ * **On-disk bulk files** in the library directory (Parquet/GeoParquet +
118
+ ``.eolas-meta.json`` sidecars) — cleared when ``files=True``.
119
+ * **Session metadata** (``Client._meta_cache`` / ``info()`` per dataset) —
120
+ cleared when ``meta=True`` and ``meta_cache`` is the client's dict.
121
+
122
+ Does not contact the API. Use :meth:`Client.get_local` with ``force=True``
123
+ to clear and re-download in one step.
124
+
125
+ Args:
126
+ name: Dataset identifier. ``None`` sweeps the whole library (``files``)
127
+ and/or all session metadata entries (``meta``).
128
+ cache_dir: Library directory. ``None`` uses :func:`resolve_library_dir`.
129
+ format: ``"parquet"``, ``"csv_gz"``, or ``"geoparquet"``. ``None``
130
+ deletes all bulk variants for ``name`` (ignored when ``name`` is
131
+ ``None``).
132
+ files: Delete on-disk bulk data files and sidecars.
133
+ meta: Drop session-cached ``info()`` entries from ``meta_cache``.
134
+ meta_cache: The client's ``_meta_cache`` dict (required for ``meta``).
135
+
136
+ Returns:
137
+ ``{"files": [deleted paths], "meta_cleared": int}``
138
+ """
139
+ deleted: list[str] = []
140
+ meta_n = 0
141
+
142
+ if meta and meta_cache is not None:
143
+ if name is None:
144
+ meta_n = len(meta_cache)
145
+ meta_cache.clear()
146
+ else:
147
+ key = str(name)
148
+ if key in meta_cache:
149
+ del meta_cache[key]
150
+ meta_n = 1
151
+
152
+ if files:
153
+ root = (
154
+ pathlib.Path(cache_dir).expanduser().resolve()
155
+ if cache_dir is not None
156
+ else resolve_library_dir(interactive=False)
157
+ )
158
+ if name is None:
159
+ if root.is_dir():
160
+ for p in root.iterdir():
161
+ if p.suffix in {".parquet", ".gz"} or p.name.endswith(".geo.parquet"):
162
+ p.unlink(missing_ok=True)
163
+ deleted.append(str(p))
164
+ elif p.name.endswith(".eolas-meta.json"):
165
+ p.unlink(missing_ok=True)
166
+ deleted.append(str(p))
167
+ elif format is None:
168
+ for ext in _BULK_EXTENSIONS.values():
169
+ p = root / f"{name}{ext}"
170
+ for candidate in (p, pathlib.Path(str(p) + ".eolas-meta.json")):
171
+ if candidate.exists():
172
+ candidate.unlink()
173
+ deleted.append(str(candidate))
174
+ else:
175
+ fmt = format.lower()
176
+ if fmt not in _BULK_EXTENSIONS:
177
+ raise ValueError(
178
+ f"Unknown format {format!r}. Expected one of: "
179
+ + ", ".join(_BULK_EXTENSIONS)
180
+ )
181
+ p = root / f"{name}{_BULK_EXTENSIONS[fmt]}"
182
+ for candidate in (p, pathlib.Path(str(p) + ".eolas-meta.json")):
183
+ if candidate.exists():
184
+ candidate.unlink()
185
+ deleted.append(str(candidate))
186
+
187
+ return {"files": deleted, "meta_cleared": meta_n}
188
+
189
+
97
190
  def library_status() -> dict:
98
191
  """Return a dict describing how the library directory is currently resolved.
99
192
 
@@ -68,6 +68,40 @@ def split_meta(info: dict) -> tuple[dict, Optional[pd.DataFrame]]:
68
68
  return table, columns
69
69
 
70
70
 
71
+ def _column_meta_records(
72
+ column_meta: Optional[pd.DataFrame | list[dict[str, Any]]],
73
+ ) -> Optional[list[dict[str, Any]]]:
74
+ if column_meta is None:
75
+ return None
76
+ if isinstance(column_meta, pd.DataFrame):
77
+ if column_meta.empty:
78
+ return None
79
+ return column_meta.to_dict("records")
80
+ if isinstance(column_meta, list):
81
+ return column_meta or None
82
+ return None
83
+
84
+
85
+ def _column_meta_dataframe(
86
+ column_meta: Optional[pd.DataFrame | list[dict[str, Any]]],
87
+ ) -> Optional[pd.DataFrame]:
88
+ if column_meta is None:
89
+ return None
90
+ if isinstance(column_meta, pd.DataFrame):
91
+ return column_meta if not column_meta.empty else None
92
+ if isinstance(column_meta, list) and column_meta:
93
+ return pd.DataFrame(column_meta)
94
+ return None
95
+
96
+
97
+ def _is_geodataframe(df: pd.DataFrame) -> bool:
98
+ try:
99
+ import geopandas as gpd
100
+ except ImportError:
101
+ return False
102
+ return isinstance(df, gpd.GeoDataFrame)
103
+
104
+
71
105
  def attach_meta(
72
106
  df: pd.DataFrame,
73
107
  *,
@@ -77,16 +111,32 @@ def attach_meta(
77
111
  column_meta: Optional[pd.DataFrame] = None,
78
112
  ) -> pd.DataFrame:
79
113
  """Attach eolas metadata attrs to a DataFrame / Dataset / GeoDataFrame."""
80
- payload = {
114
+ col_records = _column_meta_records(column_meta)
115
+ col_df = _column_meta_dataframe(column_meta)
116
+ # attrs must be JSON-serialisable — never store a DataFrame there (breaks
117
+ # pandas repr/head on GeoDataFrames when attrs are compared with ==).
118
+ attrs_payload = {
81
119
  "eolas_name": name,
82
120
  "eolas_source": source,
83
121
  "eolas_meta": table_meta or {},
84
- "eolas_columns": column_meta,
122
+ "eolas_columns": col_records,
85
123
  }
86
124
  attrs = getattr(df, "attrs", None)
87
125
  if isinstance(attrs, dict):
88
- attrs.update(payload)
89
- for key, val in payload.items():
126
+ attrs.update(attrs_payload)
127
+
128
+ if _is_geodataframe(df):
129
+ # GeoDataFrame: attrs only — setattr triggers geopandas/pandas warnings
130
+ # and can break tabular repr.
131
+ return df
132
+
133
+ object_payload = {
134
+ "eolas_name": name,
135
+ "eolas_source": source,
136
+ "eolas_meta": table_meta or {},
137
+ "eolas_columns": col_df,
138
+ }
139
+ for key, val in object_payload.items():
90
140
  try:
91
141
  setattr(df, key, val)
92
142
  except (AttributeError, TypeError):
@@ -108,7 +158,11 @@ def meta_subtitle(table_meta: Optional[dict]) -> str:
108
158
  return " · ".join(parts)
109
159
 
110
160
 
111
- def column_label(column_meta: Optional[pd.DataFrame], column: str) -> Optional[str]:
161
+ def column_label(
162
+ column_meta: Optional[pd.DataFrame | list[dict[str, Any]]],
163
+ column: str,
164
+ ) -> Optional[str]:
165
+ column_meta = _column_meta_dataframe(column_meta)
112
166
  if column_meta is None or column_meta.empty or "name" not in column_meta.columns:
113
167
  return None
114
168
  rows = column_meta.loc[column_meta["name"] == column, "description"]
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "eolas-data"
7
- version = "1.3.4"
7
+ version = "1.3.13"
8
8
  description = "Python client for the eolas.fyi statistical data API (NZ, Australia, OECD)"
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -62,6 +62,32 @@ def test_meta_false_skips_info_fetch(client):
62
62
  assert df.eolas_columns is None
63
63
 
64
64
 
65
+ def test_attach_meta_geodataframe_head_does_not_break_repr():
66
+ gpd = pytest.importorskip("geopandas")
67
+ from shapely.geometry import Point
68
+
69
+ from eolas_data.meta import attach_meta
70
+
71
+ gdf = gpd.GeoDataFrame(
72
+ {"id": [1]},
73
+ geometry=[Point(0, 0)],
74
+ crs="EPSG:4326",
75
+ )
76
+ cols = pd.DataFrame([
77
+ {"name": "id", "type": "int", "description": "Identifier"},
78
+ ])
79
+ attach_meta(
80
+ gdf,
81
+ name="nz_addresses",
82
+ source="LINZ",
83
+ table_meta={"title": "NZ Addresses"},
84
+ column_meta=cols,
85
+ )
86
+ gdf.head() # must not raise (pandas attrs must not hold DataFrames)
87
+ assert gdf.attrs["eolas_name"] == "nz_addresses"
88
+ assert gdf.attrs["eolas_columns"][0]["name"] == "id"
89
+
90
+
65
91
  @resp_lib.activate
66
92
  def test_info_cached_on_second_get(client):
67
93
  resp_lib.add(resp_lib.GET, f"{BASE}/v1/datasets/nz_cpi/data", json={"data": RECORDS})
@@ -203,6 +203,44 @@ def test_progress_kwarg_forces_hide(client, tmp_path):
203
203
  assert all(disabled_values), "tqdm must be disabled when progress=False"
204
204
 
205
205
 
206
+ # ---------------------------------------------------------------------------
207
+ # test_download_bulk_missing_content_length
208
+ # ---------------------------------------------------------------------------
209
+
210
+ @resp_lib.activate
211
+ def test_download_bulk_missing_content_length(client, tmp_path):
212
+ """CDN may omit Content-Length — download must succeed with indeterminate bar."""
213
+ resp_lib.add(resp_lib.GET, f"{BASE}/v1/datasets/nz_cpi",
214
+ json=BULK_DATASET_META, status=200)
215
+ resp_lib.add(resp_lib.GET, f"{BASE}/v1/bulk/statsnz/nz_cpi",
216
+ body=FAKE_PARQUET,
217
+ content_type="application/octet-stream",
218
+ status=200)
219
+ # Deliberately no Content-Length header.
220
+
221
+ dest = tmp_path / "nz_cpi.parquet"
222
+ totals_seen = []
223
+
224
+ class FakeTqdm:
225
+ def __init__(self, **kwargs):
226
+ totals_seen.append(kwargs.get("total"))
227
+ def __enter__(self):
228
+ return self
229
+ def __exit__(self, *a):
230
+ pass
231
+ def update(self, n):
232
+ pass
233
+
234
+ with patch("sys.stdout") as mock_stdout, \
235
+ patch("tqdm.auto.tqdm", FakeTqdm):
236
+ mock_stdout.isatty.return_value = True
237
+ out = client.download_bulk("nz_cpi", path=dest, progress=True)
238
+
239
+ assert out == dest
240
+ assert dest.read_bytes() == FAKE_PARQUET
241
+ assert totals_seen == [None]
242
+
243
+
206
244
  # ---------------------------------------------------------------------------
207
245
  # test_sync_bulk_unchanged_no_bar
208
246
  # ---------------------------------------------------------------------------
@@ -421,6 +459,25 @@ def test_eolas_no_progress_env_suppresses_bar(client, tmp_path, monkeypatch):
421
459
  assert all(disabled_values), "EOLAS_NO_PROGRESS=1 must disable tqdm even when isatty=True"
422
460
 
423
461
 
462
+ def test_progress_phase_selectors():
463
+ from eolas_data.client import Client
464
+
465
+ phases = Client._resolve_progress_phases("download")
466
+ assert phases == {"download": True, "read": False}
467
+
468
+ phases = Client._resolve_progress_phases("read")
469
+ assert phases == {"download": False, "read": True}
470
+
471
+ phases = Client._resolve_progress_phases("both")
472
+ assert phases == {"download": True, "read": True}
473
+
474
+ phases = Client._resolve_progress_phases("none")
475
+ assert phases == {"download": False, "read": False}
476
+
477
+ assert Client._resolve_show_progress("read", "read") is True
478
+ assert Client._resolve_show_progress("download", "read") is False
479
+
480
+
424
481
  def test_progress_resolver_detects_jupyter():
425
482
  """Jupyter wraps stdout so isatty() is False, but the user IS interactive
426
483
  and tqdm.auto can render a widget. Resolver must return True when
@@ -1,5 +1,7 @@
1
1
  """Live API smoke tests — run against api.eolas.fyi with a real key.
2
2
 
3
+ Mirrors eolas-r/tests/smoke-live.R and eolas/docs/client-contract.md.
4
+
3
5
  Skipped in the default unit CI job (``pytest -m 'not integration'``).
4
6
  Enable locally or in the weekly smoke workflow::
5
7
 
@@ -8,6 +10,7 @@ Enable locally or in the weekly smoke workflow::
8
10
  from __future__ import annotations
9
11
 
10
12
  import os
13
+ from pathlib import Path
11
14
 
12
15
  import pandas as pd
13
16
  import pytest
@@ -19,8 +22,17 @@ pytestmark = pytest.mark.integration
19
22
  _SKIP = not os.environ.get("EOLAS_API_KEY")
20
23
 
21
24
 
22
- @pytest.fixture(scope="module")
23
- def live_client():
25
+ @pytest.fixture()
26
+ def smoke_library(tmp_path, monkeypatch):
27
+ """Isolated on-disk library — same idea as EOLAS_LIBRARY in R smoke."""
28
+ lib = tmp_path / "eolas-smoke-lib"
29
+ lib.mkdir()
30
+ monkeypatch.setenv("EOLAS_LIBRARY", str(lib))
31
+ return lib
32
+
33
+
34
+ @pytest.fixture()
35
+ def live_client(smoke_library):
24
36
  if _SKIP:
25
37
  pytest.skip("EOLAS_API_KEY not set — live smoke tests skipped")
26
38
  return Client()
@@ -53,4 +65,21 @@ def test_get_nz_cpi_slice(live_client):
53
65
  def test_info_nz_cpi(live_client):
54
66
  meta = live_client.info("nz_cpi")
55
67
  assert meta["name"] == "nz_cpi"
56
- assert "source" in meta
68
+ assert "source" in meta
69
+
70
+
71
+ def test_client_exports_cache_clear(live_client):
72
+ assert hasattr(live_client, "cache_clear")
73
+ assert callable(live_client.cache_clear)
74
+
75
+
76
+ def test_linz_nz_addresses_bulk_route_and_head(live_client, smoke_library):
77
+ """User path: client.linz() — not get_local(). head() must not crash."""
78
+ gdf = live_client.linz("nz_addresses", progress=False)
79
+ assert len(gdf) > 100_000
80
+ # GeoDataFrame or Dataset — repr/head must work (pandas attrs regression)
81
+ gdf.head()
82
+ # Bulk file landed in isolated library
83
+ assert any(smoke_library.iterdir())
84
+ geo_files = list(smoke_library.glob("nz_addresses*.geo.parquet"))
85
+ assert geo_files, f"expected bulk geo file under {smoke_library}"
@@ -438,3 +438,62 @@ def test_cli_sync_unknown_freshness_exits_usage():
438
438
  def test_cli_sync_invalid_watch_exits_usage():
439
439
  result = runner.invoke(app, ["sync", "nz_cpi", "--watch", "forever", "--api-key", "k"])
440
440
  assert result.exit_code == cli_module.EXIT_USAGE
441
+
442
+
443
+ # ---------------------------------------------------------------------------
444
+ # cache_clear / force
445
+ # ---------------------------------------------------------------------------
446
+
447
+ @resp_lib.activate
448
+ def test_sync_bulk_force_redownloads_when_unchanged(client, tmp_path):
449
+ """force=True skips the sidecar unchanged fast path."""
450
+ dest = tmp_path / "nz_cpi.parquet"
451
+ dest.write_bytes(FAKE_PARQUET)
452
+ _write_sidecar(dest, SNAPSHOT_V1)
453
+
454
+ resp_lib.add(resp_lib.GET, f"{BASE}/v1/datasets/nz_cpi",
455
+ json=BULK_DATASET_META, status=200)
456
+ resp_lib.add(resp_lib.HEAD, f"{BASE}/v1/bulk/statsnz/nz_cpi",
457
+ body=b"", status=200,
458
+ headers={"X-Snapshot-Version": SNAPSHOT_V1})
459
+ resp_lib.add(resp_lib.GET, f"{BASE}/v1/bulk/statsnz/nz_cpi",
460
+ body=FAKE_PARQUET_V2,
461
+ content_type="application/octet-stream",
462
+ status=200)
463
+
464
+ result = client.sync_bulk("nz_cpi", path=dest, force=True)
465
+
466
+ assert result.status == "updated"
467
+ assert dest.read_bytes() == FAKE_PARQUET_V2
468
+
469
+
470
+ def test_cache_clear_removes_files_and_session_meta(client, tmp_path, monkeypatch):
471
+ """cache_clear deletes on-disk bulk files and drops session info cache."""
472
+ monkeypatch.setenv("EOLAS_LIBRARY", str(tmp_path))
473
+
474
+ dest = tmp_path / "nz_parcels.geo.parquet"
475
+ dest.write_bytes(FAKE_PARQUET)
476
+ _write_sidecar(dest, SNAPSHOT_V1)
477
+
478
+ client._meta_cache["nz_parcels"] = {"name": "nz_parcels"}
479
+
480
+ cleared = client.cache_clear("nz_parcels")
481
+
482
+ assert not dest.exists()
483
+ assert not pathlib.Path(str(dest) + ".eolas-meta.json").exists()
484
+ assert "nz_parcels" not in client._meta_cache
485
+ assert len(cleared["files"]) >= 2
486
+ assert cleared["meta_cleared"] == 1
487
+
488
+
489
+ def test_cache_clear_meta_only(client):
490
+ """files=False clears session metadata without deleting library files."""
491
+ client._meta_cache["nz_cpi"] = {"name": "nz_cpi"}
492
+ client._meta_cache["nz_parcels"] = {"name": "nz_parcels"}
493
+
494
+ cleared = client.cache_clear("nz_cpi", files=False)
495
+
496
+ assert "nz_cpi" not in client._meta_cache
497
+ assert "nz_parcels" in client._meta_cache
498
+ assert cleared["files"] == []
499
+ assert cleared["meta_cleared"] == 1
File without changes