sdrbench 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,7 @@
1
+ __pycache__/
2
+ *.egg-info/
3
+ dist/
4
+ build/
5
+ .pytest_cache/
6
+ src/sdrbench/_version.py
7
+ check.json
sdrbench-0.1.0/LICENSE ADDED
@@ -0,0 +1,32 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2026, SDRBench team
4
+
5
+ Redistribution and use in source and binary forms, with or without
6
+ modification, are permitted provided that the following conditions are met:
7
+
8
+ 1. Redistributions of source code must retain the above copyright notice, this
9
+ list of conditions and the following disclaimer.
10
+
11
+ 2. Redistributions in binary form must reproduce the above copyright notice,
12
+ this list of conditions and the following disclaimer in the documentation
13
+ and/or other materials provided with the distribution.
14
+
15
+ 3. Neither the name of the copyright holder nor the names of its
16
+ contributors may be used to endorse or promote products derived from
17
+ this software without specific prior written permission.
18
+
19
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
20
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
21
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
22
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
23
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
24
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
25
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
26
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
27
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
28
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
29
+
30
+ This license covers the code in this repository only. The datasets are
31
+ provided by their original owners under the terms stated on
32
+ https://sdrbench.github.io/ and on each Hugging Face dataset card.
@@ -0,0 +1,174 @@
1
+ Metadata-Version: 2.5
2
+ Name: sdrbench
3
+ Version: 0.1.0
4
+ Summary: Download and load SDRBench scientific datasets from Hugging Face or the original Globus archives
5
+ Project-URL: Homepage, https://sdrbench.github.io/
6
+ Project-URL: Source, https://github.com/szcompressor/sdrbench
7
+ Project-URL: Hugging Face, https://huggingface.co/sdrbench
8
+ Author: SDRBench team
9
+ License-Expression: BSD-3-Clause
10
+ License-File: LICENSE
11
+ Keywords: benchmark,hpc,hugging face,lossy compression,scientific data,sdrbench
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Scientific/Engineering
16
+ Requires-Python: >=3.9
17
+ Requires-Dist: filelock>=3.8
18
+ Requires-Dist: huggingface-hub>=0.23
19
+ Requires-Dist: numpy>=1.21
20
+ Provides-Extra: sz3
21
+ Requires-Dist: pysz>=1.1; extra == 'sz3'
22
+ Provides-Extra: test
23
+ Requires-Dist: pytest>=7; extra == 'test'
24
+ Description-Content-Type: text/markdown
25
+
26
+ # sdrbench
27
+
28
+ The [SDRBench](https://sdrbench.github.io/) scientific datasets as numpy arrays, served from the
29
+ [Hugging Face mirror](https://huggingface.co/sdrbench) (byte-for-byte copies of the original
30
+ archives, pinned and sha256-verified), with automatic fallback to the original Globus archives at
31
+ Argonne.
32
+
33
+ ```bash
34
+ pip install sdrbench # or "sdrbench[sz3]" to also get pysz (SZ3)
35
+ ```
36
+
37
+ ```python
38
+ import sdrbench
39
+
40
+ sdrbench.list() # ['basilisk-turbulence', 'cesm-atm', 'exaalt', ...]
41
+ sdrbench.list(variants=True) # ['basilisk-turbulence/small', ..., 'nyx/original', 'nyx/log', ...]
42
+
43
+ nyx = sdrbench.dataset("nyx") # default variant; sdrbench.dataset("nyx", "log")
44
+ nyx.fields # ['baryon_density', 'dark_matter_density', 'temperature', ...]
45
+ t = nyx["temperature"] # numpy memmap, float32, shape (512, 512, 512), C order
46
+ sdrbench.load("cesm-atm", "CLDHGH") # one-liner
47
+
48
+ # time series: fields are "var/step"
49
+ hur = sdrbench.dataset("hurricane-isabel", "P")
50
+ hur.steps # ['01', ..., '48']
51
+ p = hur.series("P") # ordered sequence of fields
52
+ x = hur["P", "07"] # one step
53
+ block = p.stack(slice(0, 4)) # (4, 100, 500, 500) copy
54
+
55
+ # stacked files are split into zero-copy views per variable
56
+ T = sdrbench.dataset("s3d")["T/1.1000E-03"] # (500, 500, 500), component 6 of the stored file
57
+
58
+ # raw files to a directory (original SDRBench names, downloaded in parallel)
59
+ nyx.download("data/") # -> data/512x512x512/*.f32
60
+ sdrbench.dataset("nyx", root="data/") # later: use local files first (also $SDRBENCH_DATA)
61
+
62
+ print(nyx.citation()) # BibTeX + provider request + exact data version
63
+ ```
64
+
65
+ Compress with SZ3 through [pysz](https://pypi.org/project/pysz/):
66
+
67
+ ```python
68
+ import numpy as np
69
+ from pysz import sz, szConfig, szErrorBoundMode
70
+
71
+ x = sdrbench.dataset("scale-letkf")["T"]
72
+ conf = szConfig()
73
+ conf.errorBoundMode = szErrorBoundMode.REL
74
+ conf.relErrorBound = 1e-3
75
+ compressed, ratio = sz.compress(np.ascontiguousarray(x), conf)
76
+ y, _ = sz.decompress(compressed, x.dtype.type, x.shape)
77
+ print(ratio, sz.verify(np.asarray(x), y)) # ratio, (max error, PSNR, NRMSE)
78
+ ```
79
+
80
+ Command line:
81
+
82
+ ```bash
83
+ sdrbench list # every dataset/variant with field count and size
84
+ sdrbench info hurricane-isabel/P # description, fields, dtype, shape, files
85
+ sdrbench download nyx temperature -o data/ # plain files under data/<repo path>, -j parallel
86
+ sdrbench path cesm-atm CLDHGH # local path (downloads if needed), e.g. for `sz3 -i`
87
+ sdrbench cite qmcpack
88
+ ```
89
+
90
+ ## How it works
91
+
92
+ - **Datasets, variants, fields.** A dataset (`nyx`) has variants with descriptive names (`original`,
93
+ `log`; `cesm-atm`: `2d`, `2d-cleared`, `3d`; `hacc`: `medium`, `big`, `sbig`, `region1`...;
94
+ `hurricane-isabel`: `snapshot`, `P`, `QCLOUD`, ...). The first is the default. The SDRBench / Hugging
95
+ Face folder name also works as a variant name (`dataset("nyx", "512x512x512")`).
96
+ - **Field names are physical variables** (`CLDHGH`, `T`, `temperature`), or `var/step` for time series
97
+ and slabs (`P/07`, `xx/00042`, `density/31`); `ds.variables`, `ds.steps` and `ds.series(var)` expose the
98
+ structure. The original SDRBench file name also works (`ds["CLDHGH_1_1800_3600.f32"]`), lookups are
99
+ case-insensitive, and errors suggest close matches. Non-array files are in `ds.extra_files`.
100
+ - **dtype and shape come from the catalog, never from names. Shapes are C order** (slowest first; the
101
+ convention of the SZ3 test-suite table): `np.fromfile(path, dtype).reshape(field.stored_shape)` reads
102
+ any stored file. `field.shape_fastest_first` is the order command-line compressors expect.
103
+ - **Views, no copies of the data.** QMCPACK's default variant `preconditioned` (288 x 115 x 69 x 69, the
104
+ layout of the SDRBench examples) is computed on load by transposing the stored native file (variant
105
+ `original`); S3D's stacked files are split into per-variable memmap views. `field.save(path)` writes
106
+ the array exactly as loaded.
107
+ - **Downloads** come from Hugging Face, pinned to the commit this release's catalog was built from and
108
+ checked against the catalog sha256, and are cached (`SDRBENCH_CACHE` moves the cache). Local copies are
109
+ used first when `root=` / `$SDRBENCH_DATA` is given (an untarred SDRBench directory works). If Hugging
110
+ Face cannot deliver, the package falls back to the original archive on Globus (whole archive, md5 and
111
+ sha256 verified, size shown in a warning; resumes dropped connections). `source="globus"` forces it.
112
+ - Arrays are read-only memmaps; `np.array(x)` (or `load(mmap=False)`) gives a writable copy, e.g. for
113
+ `torch.from_numpy`.
114
+
115
+ ## Datasets
116
+
117
+ | Dataset | Hugging Face | Variants | Source |
118
+ |---|---|---|---|
119
+ | CESM-ATM | [sdrbench/cesm-atm](https://huggingface.co/datasets/sdrbench/cesm-atm) | 2d, 2d-cleared, 3d | climate (SNL) |
120
+ | EXAALT | [sdrbench/exaalt](https://huggingface.co/datasets/sdrbench/exaalt) | small, copper-1/2, helium-1/2 | molecular dynamics |
121
+ | Hurricane ISABEL | [sdrbench/hurricane-isabel](https://huggingface.co/datasets/sdrbench/hurricane-isabel) | snapshot, P, U, ..., CLOUD_log10, ... | weather (NCAR, IEEE Vis 2004) |
122
+ | EXAFEL | [sdrbench/exafel](https://huggingface.co/datasets/sdrbench/exafel) | small, large, assembled | LCLS X-ray images |
123
+ | HACC | [sdrbench/hacc](https://huggingface.co/datasets/sdrbench/hacc) | medium, big, sbig, region1-6 | cosmology particles |
124
+ | NYX | [sdrbench/nyx](https://huggingface.co/datasets/sdrbench/nyx) | original, log | cosmology |
125
+ | NWChem | [sdrbench/nwchem](https://huggingface.co/datasets/sdrbench/nwchem) | default, f32 | quantum chemistry |
126
+ | SCALE-LETKF | [sdrbench/scale-letkf](https://huggingface.co/datasets/sdrbench/scale-letkf) | original, log | weather (RIKEN) |
127
+ | QMCPACK | [sdrbench/qmcpack](https://huggingface.co/datasets/sdrbench/qmcpack) | preconditioned, original | quantum Monte Carlo |
128
+ | Miranda | [sdrbench/miranda](https://huggingface.co/datasets/sdrbench/miranda) | small, big | turbulence (LLNL) |
129
+ | S3D | [sdrbench/s3d](https://huggingface.co/datasets/sdrbench/s3d) | default | combustion (SNL) |
130
+ | Basilisk-Turbulence | [sdrbench/basilisk-turbulence](https://huggingface.co/datasets/sdrbench/basilisk-turbulence) | small, large | 2D turbulence |
131
+ | XGC | [sdrbench/xgc](https://huggingface.co/datasets/sdrbench/xgc) | raw, adios | fusion (PPPL) |
132
+
133
+ `sdrbench list` and `sdrbench info` show fields and sizes. NSTX GPI is not mirrored (its owner asks to
134
+ be contacted before results are published); HACC region 5 is not available on Globus (HTTP 404).
135
+
136
+ Please cite SDRBench and the data provider named on each dataset card (`sdrbench cite <dataset>`), and
137
+ report the `sdrbench` version you used: a release always reads the same data.
138
+
139
+ ## Keeping the mirror in sync
140
+
141
+ Globus remains the source of truth. `.github/workflows/release.yml` runs weekly:
142
+
143
+ 1. `sync/sync.py check` HEADs every archive linked from the SDRBench page and compares size, ETag and
144
+ Last-Modified with `sync/state.json`; archives that appear on, or disappear from, the page open an issue.
145
+ 2. Changed archives are re-mirrored in parallel jobs (`sync.py mirror`). Archives are streamed: each file is
146
+ unpacked, hashed and committed to `sdrbench/<dataset>` in batches and then deleted, so a runner only needs
147
+ room for the largest single file (~17 GB); the job frees disk like the SZ3 CI. Files removed from an
148
+ archive are removed from the repo, but only if the whole archive arrived and at most 25% of a variant
149
+ goes (`--allow-delete` overrides). The same command works on any machine:
150
+ `HF_TOKEN=... python sync/sync.py mirror --only <dataset>`.
151
+ 3. `sync.py catalog` regenerates `src/sdrbench/catalog.json` (pinned to the new Hugging Face revisions),
152
+ `sync.py verify` checks dtypes, element counts and dimension order against the data, `sync.py cards`
153
+ refreshes the dataset cards and file tables, and a new patch version is tagged and published to PyPI.
154
+
155
+ ### Adding a dataset
156
+
157
+ 1. Add an entry to `sync/datasets.json` (see its `_comment`): title, provider, science description, and per
158
+ variant the Hugging Face `folder`, Globus `archive`, optional `select`/`exclude`, and `rules` mapping file
159
+ globs to dtype, C-order shape and `var`/`step` name templates.
160
+ 2. `python sync/sync.py mirror --only <dataset> --dry --workdir /tmp/w` downloads and hashes without uploading
161
+ (writes `/tmp/w/state.dry.json`); fix rules until `sync.py catalog` builds without errors (it refuses files
162
+ without a rule or with a size that does not match the shape).
163
+ 3. Mirror for real, then `sync.py catalog`, `sync.py verify --only <dataset>`, `sync.py cards --only <dataset>`,
164
+ and `python sync/selftest.py --only <dataset> --workdir /big/tmp` (every field through the package).
165
+
166
+ ## Development
167
+
168
+ ```bash
169
+ pip install -e ".[test,sz3]"
170
+ pytest # offline tests
171
+ pytest -m online # end-to-end against Hugging Face and Globus
172
+ python sync/sync.py verify # data-level check of every dtype/shape (network)
173
+ python sync/selftest.py --workdir /big/tmp # every field of every dataset through the package
174
+ ```
@@ -0,0 +1,149 @@
1
+ # sdrbench
2
+
3
+ The [SDRBench](https://sdrbench.github.io/) scientific datasets as numpy arrays, served from the
4
+ [Hugging Face mirror](https://huggingface.co/sdrbench) (byte-for-byte copies of the original
5
+ archives, pinned and sha256-verified), with automatic fallback to the original Globus archives at
6
+ Argonne.
7
+
8
+ ```bash
9
+ pip install sdrbench # or "sdrbench[sz3]" to also get pysz (SZ3)
10
+ ```
11
+
12
+ ```python
13
+ import sdrbench
14
+
15
+ sdrbench.list() # ['basilisk-turbulence', 'cesm-atm', 'exaalt', ...]
16
+ sdrbench.list(variants=True) # ['basilisk-turbulence/small', ..., 'nyx/original', 'nyx/log', ...]
17
+
18
+ nyx = sdrbench.dataset("nyx") # default variant; sdrbench.dataset("nyx", "log")
19
+ nyx.fields # ['baryon_density', 'dark_matter_density', 'temperature', ...]
20
+ t = nyx["temperature"] # numpy memmap, float32, shape (512, 512, 512), C order
21
+ sdrbench.load("cesm-atm", "CLDHGH") # one-liner
22
+
23
+ # time series: fields are "var/step"
24
+ hur = sdrbench.dataset("hurricane-isabel", "P")
25
+ hur.steps # ['01', ..., '48']
26
+ p = hur.series("P") # ordered sequence of fields
27
+ x = hur["P", "07"] # one step
28
+ block = p.stack(slice(0, 4)) # (4, 100, 500, 500) copy
29
+
30
+ # stacked files are split into zero-copy views per variable
31
+ T = sdrbench.dataset("s3d")["T/1.1000E-03"] # (500, 500, 500), component 6 of the stored file
32
+
33
+ # raw files to a directory (original SDRBench names, downloaded in parallel)
34
+ nyx.download("data/") # -> data/512x512x512/*.f32
35
+ sdrbench.dataset("nyx", root="data/") # later: use local files first (also $SDRBENCH_DATA)
36
+
37
+ print(nyx.citation()) # BibTeX + provider request + exact data version
38
+ ```
39
+
40
+ Compress with SZ3 through [pysz](https://pypi.org/project/pysz/):
41
+
42
+ ```python
43
+ import numpy as np
44
+ from pysz import sz, szConfig, szErrorBoundMode
45
+
46
+ x = sdrbench.dataset("scale-letkf")["T"]
47
+ conf = szConfig()
48
+ conf.errorBoundMode = szErrorBoundMode.REL
49
+ conf.relErrorBound = 1e-3
50
+ compressed, ratio = sz.compress(np.ascontiguousarray(x), conf)
51
+ y, _ = sz.decompress(compressed, x.dtype.type, x.shape)
52
+ print(ratio, sz.verify(np.asarray(x), y)) # ratio, (max error, PSNR, NRMSE)
53
+ ```
54
+
55
+ Command line:
56
+
57
+ ```bash
58
+ sdrbench list # every dataset/variant with field count and size
59
+ sdrbench info hurricane-isabel/P # description, fields, dtype, shape, files
60
+ sdrbench download nyx temperature -o data/ # plain files under data/<repo path>, -j parallel
61
+ sdrbench path cesm-atm CLDHGH # local path (downloads if needed), e.g. for `sz3 -i`
62
+ sdrbench cite qmcpack
63
+ ```
64
+
65
+ ## How it works
66
+
67
+ - **Datasets, variants, fields.** A dataset (`nyx`) has variants with descriptive names (`original`,
68
+ `log`; `cesm-atm`: `2d`, `2d-cleared`, `3d`; `hacc`: `medium`, `big`, `sbig`, `region1`...;
69
+ `hurricane-isabel`: `snapshot`, `P`, `QCLOUD`, ...). The first is the default. The SDRBench / Hugging
70
+ Face folder name also works as a variant name (`dataset("nyx", "512x512x512")`).
71
+ - **Field names are physical variables** (`CLDHGH`, `T`, `temperature`), or `var/step` for time series
72
+ and slabs (`P/07`, `xx/00042`, `density/31`); `ds.variables`, `ds.steps` and `ds.series(var)` expose the
73
+ structure. The original SDRBench file name also works (`ds["CLDHGH_1_1800_3600.f32"]`), lookups are
74
+ case-insensitive, and errors suggest close matches. Non-array files are in `ds.extra_files`.
75
+ - **dtype and shape come from the catalog, never from names. Shapes are C order** (slowest first; the
76
+ convention of the SZ3 test-suite table): `np.fromfile(path, dtype).reshape(field.stored_shape)` reads
77
+ any stored file. `field.shape_fastest_first` is the order command-line compressors expect.
78
+ - **Views, no copies of the data.** QMCPACK's default variant `preconditioned` (288 x 115 x 69 x 69, the
79
+ layout of the SDRBench examples) is computed on load by transposing the stored native file (variant
80
+ `original`); S3D's stacked files are split into per-variable memmap views. `field.save(path)` writes
81
+ the array exactly as loaded.
82
+ - **Downloads** come from Hugging Face, pinned to the commit this release's catalog was built from and
83
+ checked against the catalog sha256, and are cached (`SDRBENCH_CACHE` moves the cache). Local copies are
84
+ used first when `root=` / `$SDRBENCH_DATA` is given (an untarred SDRBench directory works). If Hugging
85
+ Face cannot deliver, the package falls back to the original archive on Globus (whole archive, md5 and
86
+ sha256 verified, size shown in a warning; resumes dropped connections). `source="globus"` forces it.
87
+ - Arrays are read-only memmaps; `np.array(x)` (or `load(mmap=False)`) gives a writable copy, e.g. for
88
+ `torch.from_numpy`.
89
+
90
+ ## Datasets
91
+
92
+ | Dataset | Hugging Face | Variants | Source |
93
+ |---|---|---|---|
94
+ | CESM-ATM | [sdrbench/cesm-atm](https://huggingface.co/datasets/sdrbench/cesm-atm) | 2d, 2d-cleared, 3d | climate (SNL) |
95
+ | EXAALT | [sdrbench/exaalt](https://huggingface.co/datasets/sdrbench/exaalt) | small, copper-1/2, helium-1/2 | molecular dynamics |
96
+ | Hurricane ISABEL | [sdrbench/hurricane-isabel](https://huggingface.co/datasets/sdrbench/hurricane-isabel) | snapshot, P, U, ..., CLOUD_log10, ... | weather (NCAR, IEEE Vis 2004) |
97
+ | EXAFEL | [sdrbench/exafel](https://huggingface.co/datasets/sdrbench/exafel) | small, large, assembled | LCLS X-ray images |
98
+ | HACC | [sdrbench/hacc](https://huggingface.co/datasets/sdrbench/hacc) | medium, big, sbig, region1-6 | cosmology particles |
99
+ | NYX | [sdrbench/nyx](https://huggingface.co/datasets/sdrbench/nyx) | original, log | cosmology |
100
+ | NWChem | [sdrbench/nwchem](https://huggingface.co/datasets/sdrbench/nwchem) | default, f32 | quantum chemistry |
101
+ | SCALE-LETKF | [sdrbench/scale-letkf](https://huggingface.co/datasets/sdrbench/scale-letkf) | original, log | weather (RIKEN) |
102
+ | QMCPACK | [sdrbench/qmcpack](https://huggingface.co/datasets/sdrbench/qmcpack) | preconditioned, original | quantum Monte Carlo |
103
+ | Miranda | [sdrbench/miranda](https://huggingface.co/datasets/sdrbench/miranda) | small, big | turbulence (LLNL) |
104
+ | S3D | [sdrbench/s3d](https://huggingface.co/datasets/sdrbench/s3d) | default | combustion (SNL) |
105
+ | Basilisk-Turbulence | [sdrbench/basilisk-turbulence](https://huggingface.co/datasets/sdrbench/basilisk-turbulence) | small, large | 2D turbulence |
106
+ | XGC | [sdrbench/xgc](https://huggingface.co/datasets/sdrbench/xgc) | raw, adios | fusion (PPPL) |
107
+
108
+ `sdrbench list` and `sdrbench info` show fields and sizes. NSTX GPI is not mirrored (its owner asks to
109
+ be contacted before results are published); HACC region 5 is not available on Globus (HTTP 404).
110
+
111
+ Please cite SDRBench and the data provider named on each dataset card (`sdrbench cite <dataset>`), and
112
+ report the `sdrbench` version you used: a release always reads the same data.
113
+
114
+ ## Keeping the mirror in sync
115
+
116
+ Globus remains the source of truth. `.github/workflows/release.yml` runs weekly:
117
+
118
+ 1. `sync/sync.py check` HEADs every archive linked from the SDRBench page and compares size, ETag and
119
+ Last-Modified with `sync/state.json`; archives that appear on, or disappear from, the page open an issue.
120
+ 2. Changed archives are re-mirrored in parallel jobs (`sync.py mirror`). Archives are streamed: each file is
121
+ unpacked, hashed and committed to `sdrbench/<dataset>` in batches and then deleted, so a runner only needs
122
+ room for the largest single file (~17 GB); the job frees disk like the SZ3 CI. Files removed from an
123
+ archive are removed from the repo, but only if the whole archive arrived and at most 25% of a variant
124
+ goes (`--allow-delete` overrides). The same command works on any machine:
125
+ `HF_TOKEN=... python sync/sync.py mirror --only <dataset>`.
126
+ 3. `sync.py catalog` regenerates `src/sdrbench/catalog.json` (pinned to the new Hugging Face revisions),
127
+ `sync.py verify` checks dtypes, element counts and dimension order against the data, `sync.py cards`
128
+ refreshes the dataset cards and file tables, and a new patch version is tagged and published to PyPI.
129
+
130
+ ### Adding a dataset
131
+
132
+ 1. Add an entry to `sync/datasets.json` (see its `_comment`): title, provider, science description, and per
133
+ variant the Hugging Face `folder`, Globus `archive`, optional `select`/`exclude`, and `rules` mapping file
134
+ globs to dtype, C-order shape and `var`/`step` name templates.
135
+ 2. `python sync/sync.py mirror --only <dataset> --dry --workdir /tmp/w` downloads and hashes without uploading
136
+ (writes `/tmp/w/state.dry.json`); fix rules until `sync.py catalog` builds without errors (it refuses files
137
+ without a rule or with a size that does not match the shape).
138
+ 3. Mirror for real, then `sync.py catalog`, `sync.py verify --only <dataset>`, `sync.py cards --only <dataset>`,
139
+ and `python sync/selftest.py --only <dataset> --workdir /big/tmp` (every field through the package).
140
+
141
+ ## Development
142
+
143
+ ```bash
144
+ pip install -e ".[test,sz3]"
145
+ pytest # offline tests
146
+ pytest -m online # end-to-end against Hugging Face and Globus
147
+ python sync/sync.py verify # data-level check of every dtype/shape (network)
148
+ python sync/selftest.py --workdir /big/tmp # every field of every dataset through the package
149
+ ```
@@ -0,0 +1,51 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.24", "hatch-vcs>=0.4"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "sdrbench"
7
+ dynamic = ["version"]
8
+ description = "Download and load SDRBench scientific datasets from Hugging Face or the original Globus archives"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "BSD-3-Clause"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "SDRBench team" }]
14
+ keywords = ["sdrbench", "scientific data", "lossy compression", "benchmark", "hpc", "hugging face"]
15
+ classifiers = [
16
+ "Programming Language :: Python :: 3",
17
+ "Intended Audience :: Science/Research",
18
+ "Topic :: Scientific/Engineering",
19
+ "Operating System :: OS Independent",
20
+ ]
21
+ dependencies = ["numpy>=1.21", "huggingface_hub>=0.23", "filelock>=3.8"]
22
+
23
+ [project.optional-dependencies]
24
+ sz3 = ["pysz>=1.1"]
25
+ test = ["pytest>=7"]
26
+
27
+ [project.urls]
28
+ Homepage = "https://sdrbench.github.io/"
29
+ Source = "https://github.com/szcompressor/sdrbench"
30
+ "Hugging Face" = "https://huggingface.co/sdrbench"
31
+
32
+ [project.scripts]
33
+ sdrbench = "sdrbench.cli:main"
34
+
35
+ [tool.hatch.version]
36
+ source = "vcs"
37
+ fallback-version = "0.0.0"
38
+
39
+ [tool.hatch.build.hooks.vcs]
40
+ version-file = "src/sdrbench/_version.py"
41
+
42
+ [tool.hatch.build.targets.wheel]
43
+ packages = ["src/sdrbench"]
44
+
45
+ [tool.hatch.build.targets.sdist]
46
+ include = ["src/sdrbench", "tests", "README.md", "LICENSE"]
47
+
48
+ [tool.pytest.ini_options]
49
+ testpaths = ["tests"]
50
+ markers = ["online: needs network access to Hugging Face and Globus"]
51
+ addopts = "-m 'not online'"
@@ -0,0 +1,23 @@
1
+ """SDRBench scientific datasets as numpy arrays.
2
+
3
+ >>> import sdrbench
4
+ >>> sdrbench.list() # datasets; sdrbench.list(variants=True) for all variants
5
+ >>> nyx = sdrbench.dataset("nyx")
6
+ >>> nyx.fields
7
+ >>> t = nyx["temperature"] # memmap, shape (512, 512, 512), C order
8
+ >>> p = sdrbench.dataset("hurricane-isabel", "P").series("P") # 48 time steps
9
+
10
+ Files come from the Hugging Face mirror (https://huggingface.co/sdrbench), pinned to the
11
+ revision this release was built from; if Hugging Face cannot deliver, from the original
12
+ SDRBench archives on Globus. Local copies can be used via ``root=`` or ``$SDRBENCH_DATA``.
13
+ """
14
+ from ._core import Dataset, Field, Series, cache_dir, catalog, dataset, list, load
15
+
16
+ try:
17
+ from ._version import __version__
18
+ except ImportError: # source checkout without a build
19
+ __version__ = "0.0.0"
20
+
21
+ # `list` is deliberately not exported by `from sdrbench import *` (it would shadow the builtin);
22
+ # use sdrbench.list().
23
+ __all__ = ["Dataset", "Field", "Series", "cache_dir", "catalog", "dataset", "load"]
@@ -0,0 +1,219 @@
1
+ """Download and unpack SDRBench archives.
2
+
3
+ The Hugging Face mirror is built with exactly this code, so a file fetched
4
+ from Globus ends up at the same relative path, with the same bytes, as the
5
+ file stored on Hugging Face.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import hashlib
10
+ import http.client
11
+ import os
12
+ import shutil
13
+ import tarfile
14
+ import time
15
+ import urllib.error
16
+ import urllib.request
17
+ import zipfile
18
+ from pathlib import Path
19
+ from typing import Optional, Tuple
20
+
21
+ CHUNK = 1 << 22
22
+ JUNK = ("._", ".DS_Store", "__MACOSX")
23
+
24
+
25
+ def _is_junk(name: str) -> bool:
26
+ return any(part.startswith(JUNK) for part in Path(name).parts)
27
+
28
+
29
+ def sha256_file(path, chunk: int = 1 << 24) -> str:
30
+ h = hashlib.sha256()
31
+ with open(path, "rb") as f:
32
+ while b := f.read(chunk):
33
+ h.update(b)
34
+ return h.hexdigest()
35
+
36
+
37
+ class _HashingReader:
38
+ """File-like wrapper that hashes and counts everything read through it."""
39
+
40
+ def __init__(self, raw):
41
+ self.raw = raw
42
+ self.md5 = hashlib.md5()
43
+ self.nbytes = 0
44
+
45
+ def read(self, n: int = -1) -> bytes:
46
+ b = self.raw.read(n)
47
+ self.md5.update(b)
48
+ self.nbytes += len(b)
49
+ return b
50
+
51
+ def drain(self) -> None:
52
+ while self.read(CHUNK):
53
+ pass
54
+
55
+
56
+ class _ResumingResponse:
57
+ """Readable HTTP body that reconnects with a Range request when the connection drops,
58
+ so multi-GB downloads survive transient network failures.
59
+
60
+ The full size must be known (Globus sends GET bodies chunked, without Content-Length, so it
61
+ comes from HEAD); without it a dropped connection cannot be told apart from the end of the
62
+ file, so we refuse to download instead of risking a silently truncated archive."""
63
+
64
+ def __init__(self, url: str, retries: int = 10, timeout: float = 120):
65
+ self.url, self.retries, self.timeout = url, retries, timeout
66
+ self.pos = 0
67
+ self.total = self._head_length()
68
+ self.resp = self._connect()
69
+
70
+ def _head_length(self) -> int:
71
+ last = None
72
+ for attempt in range(self.retries + 1):
73
+ req = urllib.request.Request(self.url, method="HEAD", headers={"User-Agent": "sdrbench"})
74
+ try:
75
+ with urllib.request.urlopen(req, timeout=self.timeout) as r:
76
+ n = r.headers.get("Content-Length")
77
+ if n is not None:
78
+ return int(n)
79
+ last = "no Content-Length in HEAD response"
80
+ except urllib.error.HTTPError as e:
81
+ if e.code not in (429, 502, 503, 504): # only temporary server errors are retried
82
+ raise
83
+ last = e
84
+ except (OSError, http.client.HTTPException, ValueError) as e:
85
+ last = e
86
+ time.sleep(min(60, 2 ** attempt))
87
+ raise IOError(f"{self.url}: cannot determine the archive size ({last}); refusing an unverifiable download")
88
+
89
+ def _connect(self):
90
+ headers = {"User-Agent": "sdrbench"}
91
+ if self.pos:
92
+ headers["Range"] = f"bytes={self.pos}-"
93
+ resp = urllib.request.urlopen(urllib.request.Request(self.url, headers=headers), timeout=self.timeout)
94
+ if self.pos:
95
+ rng = resp.headers.get("Content-Range", "")
96
+ if resp.status != 206 or not rng.startswith(f"bytes {self.pos}-"):
97
+ resp.close()
98
+ raise IOError(f"{self.url}: server did not resume at byte {self.pos} ({resp.status} {rng!r})")
99
+ return resp
100
+
101
+ def read(self, n: int = -1) -> bytes:
102
+ for attempt in range(self.retries + 1):
103
+ try:
104
+ b = self.resp.read(n)
105
+ if not b and n != 0 and self.pos < self.total:
106
+ # body ended early: a dropped connection (with chunked encoding this can look
107
+ # like a clean end of stream)
108
+ raise http.client.IncompleteRead(b"", self.total - self.pos)
109
+ self.pos += len(b)
110
+ if self.pos > self.total:
111
+ raise IOError(f"{self.url}: received more than the {self.total} bytes announced")
112
+ return b
113
+ except (OSError, http.client.HTTPException) as e:
114
+ if isinstance(e, IOError) and "announced" in str(e) or attempt == self.retries:
115
+ raise
116
+ time.sleep(min(60, 2 ** attempt))
117
+ try:
118
+ self.resp.close()
119
+ except Exception:
120
+ pass
121
+ try:
122
+ self.resp = self._connect()
123
+ except (OSError, http.client.HTTPException):
124
+ continue # try again on the next attempt
125
+ raise IOError(f"{self.url}: download failed at byte {self.pos} of {self.total}")
126
+
127
+ def close(self):
128
+ self.resp.close()
129
+
130
+ def __enter__(self):
131
+ return self
132
+
133
+ def __exit__(self, *exc):
134
+ self.close()
135
+
136
+
137
+ def _open_url(url: str):
138
+ if url.startswith(("http://", "https://")):
139
+ return _ResumingResponse(url)
140
+ return urllib.request.urlopen(urllib.request.Request(url, headers={"User-Agent": "sdrbench"}), timeout=120)
141
+
142
+
143
+ def _safe_target(root: Path, name: str) -> Path:
144
+ target = (root / name).resolve()
145
+ if os.path.commonpath([str(root.resolve()), str(target)]) != str(root.resolve()):
146
+ raise ValueError(f"unsafe path in archive: {name}")
147
+ return target
148
+
149
+
150
+ def _extract_tar_stream(stream, root: Path) -> None:
151
+ with tarfile.open(fileobj=stream, mode="r|*") as tf:
152
+ for m in tf:
153
+ if _is_junk(m.name) or not (m.isfile() or m.isdir()):
154
+ continue
155
+ target = _safe_target(root, m.name)
156
+ if m.isdir():
157
+ target.mkdir(parents=True, exist_ok=True)
158
+ continue
159
+ target.parent.mkdir(parents=True, exist_ok=True)
160
+ src = tf.extractfile(m)
161
+ with open(target, "wb") as out:
162
+ shutil.copyfileobj(src, out, CHUNK)
163
+
164
+
165
+ def _extract_zip(path: Path, root: Path) -> None:
166
+ with zipfile.ZipFile(path) as zf:
167
+ for info in zf.infolist():
168
+ if _is_junk(info.filename) or info.is_dir():
169
+ continue
170
+ target = _safe_target(root, info.filename)
171
+ target.parent.mkdir(parents=True, exist_ok=True)
172
+ with zf.open(info) as src, open(target, "wb") as out:
173
+ shutil.copyfileobj(src, out, CHUNK)
174
+
175
+
176
+ def fetch_archive(url: str, dest, expected_md5: Optional[str] = None) -> Tuple[str, int]:
177
+ """Stream ``url`` (a .tar.gz or .zip) into directory ``dest``.
178
+
179
+ If all files of the archive live under a single top-level directory, its contents are
180
+ placed directly in ``dest`` (the same rule the Hugging Face mirror uses). File contents
181
+ are never modified. The whole archive must arrive: its size is checked against the
182
+ server's, and its md5 against ``expected_md5`` when given.
183
+ Returns ``(md5 of the archive, archive size in bytes)``.
184
+ """
185
+ import tempfile
186
+
187
+ dest = Path(dest)
188
+ dest.parent.mkdir(parents=True, exist_ok=True)
189
+ tmp = Path(tempfile.mkdtemp(prefix=dest.name + ".", suffix=".partial", dir=dest.parent))
190
+ try:
191
+ with _open_url(url) as resp:
192
+ reader = _HashingReader(resp)
193
+ if url.endswith(".zip"):
194
+ zpath = tmp / "_archive.zip"
195
+ with open(zpath, "wb") as out:
196
+ shutil.copyfileobj(reader, out, CHUNK)
197
+ files = tmp / "files"
198
+ _extract_zip(zpath, files)
199
+ zpath.unlink()
200
+ else:
201
+ files = tmp / "files"
202
+ files.mkdir()
203
+ _extract_tar_stream(reader, files)
204
+ reader.drain()
205
+ total = getattr(resp, "total", None)
206
+ if total is not None and reader.nbytes != total:
207
+ raise IOError(f"{url}: got {reader.nbytes} of {total} bytes")
208
+ md5 = reader.md5.hexdigest()
209
+ if expected_md5 and md5 != expected_md5:
210
+ raise IOError(f"md5 mismatch for {url}: got {md5}, expected {expected_md5}")
211
+ tops = {p.relative_to(files).parts[0] for p in files.rglob("*") if p.is_file()}
212
+ nested = {p.relative_to(files).parts[0] for p in files.rglob("*") if p.is_file()
213
+ and len(p.relative_to(files).parts) > 1}
214
+ src = files / next(iter(tops)) if len(tops) == 1 and tops == nested else files
215
+ shutil.rmtree(dest, ignore_errors=True)
216
+ os.replace(src, dest)
217
+ return md5, reader.nbytes
218
+ finally:
219
+ shutil.rmtree(tmp, ignore_errors=True)