sdrbench 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sdrbench-0.1.0/.gitignore +7 -0
- sdrbench-0.1.0/LICENSE +32 -0
- sdrbench-0.1.0/PKG-INFO +174 -0
- sdrbench-0.1.0/README.md +149 -0
- sdrbench-0.1.0/pyproject.toml +51 -0
- sdrbench-0.1.0/src/sdrbench/__init__.py +23 -0
- sdrbench-0.1.0/src/sdrbench/_archive.py +219 -0
- sdrbench-0.1.0/src/sdrbench/_core.py +571 -0
- sdrbench-0.1.0/src/sdrbench/_version.py +24 -0
- sdrbench-0.1.0/src/sdrbench/catalog.json +39137 -0
- sdrbench-0.1.0/src/sdrbench/cli.py +69 -0
- sdrbench-0.1.0/src/sdrbench/py.typed +0 -0
- sdrbench-0.1.0/tests/conftest.py +93 -0
- sdrbench-0.1.0/tests/test_archive.py +213 -0
- sdrbench-0.1.0/tests/test_catalog.py +265 -0
- sdrbench-0.1.0/tests/test_core.py +300 -0
- sdrbench-0.1.0/tests/test_online.py +66 -0
sdrbench-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, SDRBench team
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
22
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
23
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
24
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
25
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
26
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
27
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
28
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
29
|
+
|
|
30
|
+
This license covers the code in this repository only. The datasets are
|
|
31
|
+
provided by their original owners under the terms stated on
|
|
32
|
+
https://sdrbench.github.io/ and on each Hugging Face dataset card.
|
sdrbench-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: sdrbench
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Download and load SDRBench scientific datasets from Hugging Face or the original Globus archives
|
|
5
|
+
Project-URL: Homepage, https://sdrbench.github.io/
|
|
6
|
+
Project-URL: Source, https://github.com/szcompressor/sdrbench
|
|
7
|
+
Project-URL: Hugging Face, https://huggingface.co/sdrbench
|
|
8
|
+
Author: SDRBench team
|
|
9
|
+
License-Expression: BSD-3-Clause
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: benchmark,hpc,hugging face,lossy compression,scientific data,sdrbench
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Requires-Dist: filelock>=3.8
|
|
18
|
+
Requires-Dist: huggingface-hub>=0.23
|
|
19
|
+
Requires-Dist: numpy>=1.21
|
|
20
|
+
Provides-Extra: sz3
|
|
21
|
+
Requires-Dist: pysz>=1.1; extra == 'sz3'
|
|
22
|
+
Provides-Extra: test
|
|
23
|
+
Requires-Dist: pytest>=7; extra == 'test'
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# sdrbench
|
|
27
|
+
|
|
28
|
+
The [SDRBench](https://sdrbench.github.io/) scientific datasets as numpy arrays, served from the
|
|
29
|
+
[Hugging Face mirror](https://huggingface.co/sdrbench) (byte-for-byte copies of the original
|
|
30
|
+
archives, pinned and sha256-verified), with automatic fallback to the original Globus archives at
|
|
31
|
+
Argonne.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install sdrbench # or "sdrbench[sz3]" to also get pysz (SZ3)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
import sdrbench
|
|
39
|
+
|
|
40
|
+
sdrbench.list() # ['basilisk-turbulence', 'cesm-atm', 'exaalt', ...]
|
|
41
|
+
sdrbench.list(variants=True) # ['basilisk-turbulence/small', ..., 'nyx/original', 'nyx/log', ...]
|
|
42
|
+
|
|
43
|
+
nyx = sdrbench.dataset("nyx") # default variant; sdrbench.dataset("nyx", "log")
|
|
44
|
+
nyx.fields # ['baryon_density', 'dark_matter_density', 'temperature', ...]
|
|
45
|
+
t = nyx["temperature"] # numpy memmap, float32, shape (512, 512, 512), C order
|
|
46
|
+
sdrbench.load("cesm-atm", "CLDHGH") # one-liner
|
|
47
|
+
|
|
48
|
+
# time series: fields are "var/step"
|
|
49
|
+
hur = sdrbench.dataset("hurricane-isabel", "P")
|
|
50
|
+
hur.steps # ['01', ..., '48']
|
|
51
|
+
p = hur.series("P") # ordered sequence of fields
|
|
52
|
+
x = hur["P", "07"] # one step
|
|
53
|
+
block = p.stack(slice(0, 4)) # (4, 100, 500, 500) copy
|
|
54
|
+
|
|
55
|
+
# stacked files are split into zero-copy views per variable
|
|
56
|
+
T = sdrbench.dataset("s3d")["T/1.1000E-03"] # (500, 500, 500), component 6 of the stored file
|
|
57
|
+
|
|
58
|
+
# raw files to a directory (original SDRBench names, downloaded in parallel)
|
|
59
|
+
nyx.download("data/") # -> data/512x512x512/*.f32
|
|
60
|
+
sdrbench.dataset("nyx", root="data/") # later: use local files first (also $SDRBENCH_DATA)
|
|
61
|
+
|
|
62
|
+
print(nyx.citation()) # BibTeX + provider request + exact data version
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Compress with SZ3 through [pysz](https://pypi.org/project/pysz/):
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
import numpy as np
|
|
69
|
+
from pysz import sz, szConfig, szErrorBoundMode
|
|
70
|
+
|
|
71
|
+
x = sdrbench.dataset("scale-letkf")["T"]
|
|
72
|
+
conf = szConfig()
|
|
73
|
+
conf.errorBoundMode = szErrorBoundMode.REL
|
|
74
|
+
conf.relErrorBound = 1e-3
|
|
75
|
+
compressed, ratio = sz.compress(np.ascontiguousarray(x), conf)
|
|
76
|
+
y, _ = sz.decompress(compressed, x.dtype.type, x.shape)
|
|
77
|
+
print(ratio, sz.verify(np.asarray(x), y)) # ratio, (max error, PSNR, NRMSE)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Command line:
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
sdrbench list # every dataset/variant with field count and size
|
|
84
|
+
sdrbench info hurricane-isabel/P # description, fields, dtype, shape, files
|
|
85
|
+
sdrbench download nyx temperature -o data/ # plain files under data/<repo path>, -j parallel
|
|
86
|
+
sdrbench path cesm-atm CLDHGH # local path (downloads if needed), e.g. for `sz3 -i`
|
|
87
|
+
sdrbench cite qmcpack
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## How it works
|
|
91
|
+
|
|
92
|
+
- **Datasets, variants, fields.** A dataset (`nyx`) has variants with descriptive names (`original`,
|
|
93
|
+
`log`; `cesm-atm`: `2d`, `2d-cleared`, `3d`; `hacc`: `medium`, `big`, `sbig`, `region1`...;
|
|
94
|
+
`hurricane-isabel`: `snapshot`, `P`, `QCLOUD`, ...). The first is the default. The SDRBench / Hugging
|
|
95
|
+
Face folder name also works as a variant name (`dataset("nyx", "512x512x512")`).
|
|
96
|
+
- **Field names are physical variables** (`CLDHGH`, `T`, `temperature`), or `var/step` for time series
|
|
97
|
+
and slabs (`P/07`, `xx/00042`, `density/31`); `ds.variables`, `ds.steps` and `ds.series(var)` expose the
|
|
98
|
+
structure. The original SDRBench file name also works (`ds["CLDHGH_1_1800_3600.f32"]`), lookups are
|
|
99
|
+
case-insensitive, and errors suggest close matches. Non-array files are in `ds.extra_files`.
|
|
100
|
+
- **dtype and shape come from the catalog, never from names. Shapes are C order** (slowest first; the
|
|
101
|
+
convention of the SZ3 test-suite table): `np.fromfile(path, dtype).reshape(field.stored_shape)` reads
|
|
102
|
+
any stored file. `field.shape_fastest_first` is the order command-line compressors expect.
|
|
103
|
+
- **Views, no copies of the data.** QMCPACK's default variant `preconditioned` (288 x 115 x 69 x 69, the
|
|
104
|
+
layout of the SDRBench examples) is computed on load by transposing the stored native file (variant
|
|
105
|
+
`original`); S3D's stacked files are split into per-variable memmap views. `field.save(path)` writes
|
|
106
|
+
the array exactly as loaded.
|
|
107
|
+
- **Downloads** come from Hugging Face, pinned to the commit this release's catalog was built from and
|
|
108
|
+
checked against the catalog sha256, and are cached (`SDRBENCH_CACHE` moves the cache). Local copies are
|
|
109
|
+
used first when `root=` / `$SDRBENCH_DATA` is given (an untarred SDRBench directory works). If Hugging
|
|
110
|
+
Face cannot deliver, the package falls back to the original archive on Globus (whole archive, md5 and
|
|
111
|
+
sha256 verified, size shown in a warning; resumes dropped connections). `source="globus"` forces it.
|
|
112
|
+
- Arrays are read-only memmaps; `np.array(x)` (or `load(mmap=False)`) gives a writable copy, e.g. for
|
|
113
|
+
`torch.from_numpy`.
|
|
114
|
+
|
|
115
|
+
## Datasets
|
|
116
|
+
|
|
117
|
+
| Dataset | Hugging Face | Variants | Source |
|
|
118
|
+
|---|---|---|---|
|
|
119
|
+
| CESM-ATM | [sdrbench/cesm-atm](https://huggingface.co/datasets/sdrbench/cesm-atm) | 2d, 2d-cleared, 3d | climate (SNL) |
|
|
120
|
+
| EXAALT | [sdrbench/exaalt](https://huggingface.co/datasets/sdrbench/exaalt) | small, copper-1/2, helium-1/2 | molecular dynamics |
|
|
121
|
+
| Hurricane ISABEL | [sdrbench/hurricane-isabel](https://huggingface.co/datasets/sdrbench/hurricane-isabel) | snapshot, P, U, ..., CLOUD_log10, ... | weather (NCAR, IEEE Vis 2004) |
|
|
122
|
+
| EXAFEL | [sdrbench/exafel](https://huggingface.co/datasets/sdrbench/exafel) | small, large, assembled | LCLS X-ray images |
|
|
123
|
+
| HACC | [sdrbench/hacc](https://huggingface.co/datasets/sdrbench/hacc) | medium, big, sbig, region1-6 | cosmology particles |
|
|
124
|
+
| NYX | [sdrbench/nyx](https://huggingface.co/datasets/sdrbench/nyx) | original, log | cosmology |
|
|
125
|
+
| NWChem | [sdrbench/nwchem](https://huggingface.co/datasets/sdrbench/nwchem) | default, f32 | quantum chemistry |
|
|
126
|
+
| SCALE-LETKF | [sdrbench/scale-letkf](https://huggingface.co/datasets/sdrbench/scale-letkf) | original, log | weather (RIKEN) |
|
|
127
|
+
| QMCPACK | [sdrbench/qmcpack](https://huggingface.co/datasets/sdrbench/qmcpack) | preconditioned, original | quantum Monte Carlo |
|
|
128
|
+
| Miranda | [sdrbench/miranda](https://huggingface.co/datasets/sdrbench/miranda) | small, big | turbulence (LLNL) |
|
|
129
|
+
| S3D | [sdrbench/s3d](https://huggingface.co/datasets/sdrbench/s3d) | default | combustion (SNL) |
|
|
130
|
+
| Basilisk-Turbulence | [sdrbench/basilisk-turbulence](https://huggingface.co/datasets/sdrbench/basilisk-turbulence) | small, large | 2D turbulence |
|
|
131
|
+
| XGC | [sdrbench/xgc](https://huggingface.co/datasets/sdrbench/xgc) | raw, adios | fusion (PPPL) |
|
|
132
|
+
|
|
133
|
+
`sdrbench list` and `sdrbench info` show fields and sizes. NSTX GPI is not mirrored (its owner asks to
|
|
134
|
+
be contacted before results are published); HACC region 5 is not available on Globus (HTTP 404).
|
|
135
|
+
|
|
136
|
+
Please cite SDRBench and the data provider named on each dataset card (`sdrbench cite <dataset>`), and
|
|
137
|
+
report the `sdrbench` version you used: a release always reads the same data.
|
|
138
|
+
|
|
139
|
+
## Keeping the mirror in sync
|
|
140
|
+
|
|
141
|
+
Globus remains the source of truth. `.github/workflows/release.yml` runs weekly:
|
|
142
|
+
|
|
143
|
+
1. `sync/sync.py check` HEADs every archive linked from the SDRBench page and compares size, ETag and
|
|
144
|
+
Last-Modified with `sync/state.json`; archives that appear on, or disappear from, the page open an issue.
|
|
145
|
+
2. Changed archives are re-mirrored in parallel jobs (`sync.py mirror`). Archives are streamed: each file is
|
|
146
|
+
unpacked, hashed and committed to `sdrbench/<dataset>` in batches and then deleted, so a runner only needs
|
|
147
|
+
room for the largest single file (~17 GB); the job frees disk like the SZ3 CI. Files removed from an
|
|
148
|
+
archive are removed from the repo, but only if the whole archive arrived and at most 25% of a variant
|
|
149
|
+
goes (`--allow-delete` overrides). The same command works on any machine:
|
|
150
|
+
`HF_TOKEN=... python sync/sync.py mirror --only <dataset>`.
|
|
151
|
+
3. `sync.py catalog` regenerates `src/sdrbench/catalog.json` (pinned to the new Hugging Face revisions),
|
|
152
|
+
`sync.py verify` checks dtypes, element counts and dimension order against the data, `sync.py cards`
|
|
153
|
+
refreshes the dataset cards and file tables, and a new patch version is tagged and published to PyPI.
|
|
154
|
+
|
|
155
|
+
### Adding a dataset
|
|
156
|
+
|
|
157
|
+
1. Add an entry to `sync/datasets.json` (see its `_comment`): title, provider, science description, and per
|
|
158
|
+
variant the Hugging Face `folder`, Globus `archive`, optional `select`/`exclude`, and `rules` mapping file
|
|
159
|
+
globs to dtype, C-order shape and `var`/`step` name templates.
|
|
160
|
+
2. `python sync/sync.py mirror --only <dataset> --dry --workdir /tmp/w` downloads and hashes without uploading
|
|
161
|
+
(writes `/tmp/w/state.dry.json`); fix rules until `sync.py catalog` builds without errors (it refuses files
|
|
162
|
+
without a rule or with a size that does not match the shape).
|
|
163
|
+
3. Mirror for real, then `sync.py catalog`, `sync.py verify --only <dataset>`, `sync.py cards --only <dataset>`,
|
|
164
|
+
and `python sync/selftest.py --only <dataset> --workdir /big/tmp` (every field through the package).
|
|
165
|
+
|
|
166
|
+
## Development
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
pip install -e ".[test,sz3]"
|
|
170
|
+
pytest # offline tests
|
|
171
|
+
pytest -m online # end-to-end against Hugging Face and Globus
|
|
172
|
+
python sync/sync.py verify # data-level check of every dtype/shape (network)
|
|
173
|
+
python sync/selftest.py --workdir /big/tmp # every field of every dataset through the package
|
|
174
|
+
```
|
sdrbench-0.1.0/README.md
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# sdrbench
|
|
2
|
+
|
|
3
|
+
The [SDRBench](https://sdrbench.github.io/) scientific datasets as numpy arrays, served from the
|
|
4
|
+
[Hugging Face mirror](https://huggingface.co/sdrbench) (byte-for-byte copies of the original
|
|
5
|
+
archives, pinned and sha256-verified), with automatic fallback to the original Globus archives at
|
|
6
|
+
Argonne.
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
pip install sdrbench # or "sdrbench[sz3]" to also get pysz (SZ3)
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
import sdrbench
|
|
14
|
+
|
|
15
|
+
sdrbench.list() # ['basilisk-turbulence', 'cesm-atm', 'exaalt', ...]
|
|
16
|
+
sdrbench.list(variants=True) # ['basilisk-turbulence/small', ..., 'nyx/original', 'nyx/log', ...]
|
|
17
|
+
|
|
18
|
+
nyx = sdrbench.dataset("nyx") # default variant; sdrbench.dataset("nyx", "log")
|
|
19
|
+
nyx.fields # ['baryon_density', 'dark_matter_density', 'temperature', ...]
|
|
20
|
+
t = nyx["temperature"] # numpy memmap, float32, shape (512, 512, 512), C order
|
|
21
|
+
sdrbench.load("cesm-atm", "CLDHGH") # one-liner
|
|
22
|
+
|
|
23
|
+
# time series: fields are "var/step"
|
|
24
|
+
hur = sdrbench.dataset("hurricane-isabel", "P")
|
|
25
|
+
hur.steps # ['01', ..., '48']
|
|
26
|
+
p = hur.series("P") # ordered sequence of fields
|
|
27
|
+
x = hur["P", "07"] # one step
|
|
28
|
+
block = p.stack(slice(0, 4)) # (4, 100, 500, 500) copy
|
|
29
|
+
|
|
30
|
+
# stacked files are split into zero-copy views per variable
|
|
31
|
+
T = sdrbench.dataset("s3d")["T/1.1000E-03"] # (500, 500, 500), component 6 of the stored file
|
|
32
|
+
|
|
33
|
+
# raw files to a directory (original SDRBench names, downloaded in parallel)
|
|
34
|
+
nyx.download("data/") # -> data/512x512x512/*.f32
|
|
35
|
+
sdrbench.dataset("nyx", root="data/") # later: use local files first (also $SDRBENCH_DATA)
|
|
36
|
+
|
|
37
|
+
print(nyx.citation()) # BibTeX + provider request + exact data version
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Compress with SZ3 through [pysz](https://pypi.org/project/pysz/):
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
import numpy as np
|
|
44
|
+
from pysz import sz, szConfig, szErrorBoundMode
|
|
45
|
+
|
|
46
|
+
x = sdrbench.dataset("scale-letkf")["T"]
|
|
47
|
+
conf = szConfig()
|
|
48
|
+
conf.errorBoundMode = szErrorBoundMode.REL
|
|
49
|
+
conf.relErrorBound = 1e-3
|
|
50
|
+
compressed, ratio = sz.compress(np.ascontiguousarray(x), conf)
|
|
51
|
+
y, _ = sz.decompress(compressed, x.dtype.type, x.shape)
|
|
52
|
+
print(ratio, sz.verify(np.asarray(x), y)) # ratio, (max error, PSNR, NRMSE)
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Command line:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
sdrbench list # every dataset/variant with field count and size
|
|
59
|
+
sdrbench info hurricane-isabel/P # description, fields, dtype, shape, files
|
|
60
|
+
sdrbench download nyx temperature -o data/ # plain files under data/<repo path>, -j parallel
|
|
61
|
+
sdrbench path cesm-atm CLDHGH # local path (downloads if needed), e.g. for `sz3 -i`
|
|
62
|
+
sdrbench cite qmcpack
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
## How it works
|
|
66
|
+
|
|
67
|
+
- **Datasets, variants, fields.** A dataset (`nyx`) has variants with descriptive names (`original`,
|
|
68
|
+
`log`; `cesm-atm`: `2d`, `2d-cleared`, `3d`; `hacc`: `medium`, `big`, `sbig`, `region1`...;
|
|
69
|
+
`hurricane-isabel`: `snapshot`, `P`, `QCLOUD`, ...). The first is the default. The SDRBench / Hugging
|
|
70
|
+
Face folder name also works as a variant name (`dataset("nyx", "512x512x512")`).
|
|
71
|
+
- **Field names are physical variables** (`CLDHGH`, `T`, `temperature`), or `var/step` for time series
|
|
72
|
+
and slabs (`P/07`, `xx/00042`, `density/31`); `ds.variables`, `ds.steps` and `ds.series(var)` expose the
|
|
73
|
+
structure. The original SDRBench file name also works (`ds["CLDHGH_1_1800_3600.f32"]`), lookups are
|
|
74
|
+
case-insensitive, and errors suggest close matches. Non-array files are in `ds.extra_files`.
|
|
75
|
+
- **dtype and shape come from the catalog, never from names. Shapes are C order** (slowest first; the
|
|
76
|
+
convention of the SZ3 test-suite table): `np.fromfile(path, dtype).reshape(field.stored_shape)` reads
|
|
77
|
+
any stored file. `field.shape_fastest_first` is the order command-line compressors expect.
|
|
78
|
+
- **Views, no copies of the data.** QMCPACK's default variant `preconditioned` (288 x 115 x 69 x 69, the
|
|
79
|
+
layout of the SDRBench examples) is computed on load by transposing the stored native file (variant
|
|
80
|
+
`original`); S3D's stacked files are split into per-variable memmap views. `field.save(path)` writes
|
|
81
|
+
the array exactly as loaded.
|
|
82
|
+
- **Downloads** come from Hugging Face, pinned to the commit this release's catalog was built from and
|
|
83
|
+
checked against the catalog sha256, and are cached (`SDRBENCH_CACHE` moves the cache). Local copies are
|
|
84
|
+
used first when `root=` / `$SDRBENCH_DATA` is given (an untarred SDRBench directory works). If Hugging
|
|
85
|
+
Face cannot deliver, the package falls back to the original archive on Globus (whole archive, md5 and
|
|
86
|
+
sha256 verified, size shown in a warning; resumes dropped connections). `source="globus"` forces it.
|
|
87
|
+
- Arrays are read-only memmaps; `np.array(x)` (or `load(mmap=False)`) gives a writable copy, e.g. for
|
|
88
|
+
`torch.from_numpy`.
|
|
89
|
+
|
|
90
|
+
## Datasets
|
|
91
|
+
|
|
92
|
+
| Dataset | Hugging Face | Variants | Source |
|
|
93
|
+
|---|---|---|---|
|
|
94
|
+
| CESM-ATM | [sdrbench/cesm-atm](https://huggingface.co/datasets/sdrbench/cesm-atm) | 2d, 2d-cleared, 3d | climate (SNL) |
|
|
95
|
+
| EXAALT | [sdrbench/exaalt](https://huggingface.co/datasets/sdrbench/exaalt) | small, copper-1/2, helium-1/2 | molecular dynamics |
|
|
96
|
+
| Hurricane ISABEL | [sdrbench/hurricane-isabel](https://huggingface.co/datasets/sdrbench/hurricane-isabel) | snapshot, P, U, ..., CLOUD_log10, ... | weather (NCAR, IEEE Vis 2004) |
|
|
97
|
+
| EXAFEL | [sdrbench/exafel](https://huggingface.co/datasets/sdrbench/exafel) | small, large, assembled | LCLS X-ray images |
|
|
98
|
+
| HACC | [sdrbench/hacc](https://huggingface.co/datasets/sdrbench/hacc) | medium, big, sbig, region1-6 | cosmology particles |
|
|
99
|
+
| NYX | [sdrbench/nyx](https://huggingface.co/datasets/sdrbench/nyx) | original, log | cosmology |
|
|
100
|
+
| NWChem | [sdrbench/nwchem](https://huggingface.co/datasets/sdrbench/nwchem) | default, f32 | quantum chemistry |
|
|
101
|
+
| SCALE-LETKF | [sdrbench/scale-letkf](https://huggingface.co/datasets/sdrbench/scale-letkf) | original, log | weather (RIKEN) |
|
|
102
|
+
| QMCPACK | [sdrbench/qmcpack](https://huggingface.co/datasets/sdrbench/qmcpack) | preconditioned, original | quantum Monte Carlo |
|
|
103
|
+
| Miranda | [sdrbench/miranda](https://huggingface.co/datasets/sdrbench/miranda) | small, big | turbulence (LLNL) |
|
|
104
|
+
| S3D | [sdrbench/s3d](https://huggingface.co/datasets/sdrbench/s3d) | default | combustion (SNL) |
|
|
105
|
+
| Basilisk-Turbulence | [sdrbench/basilisk-turbulence](https://huggingface.co/datasets/sdrbench/basilisk-turbulence) | small, large | 2D turbulence |
|
|
106
|
+
| XGC | [sdrbench/xgc](https://huggingface.co/datasets/sdrbench/xgc) | raw, adios | fusion (PPPL) |
|
|
107
|
+
|
|
108
|
+
`sdrbench list` and `sdrbench info` show fields and sizes. NSTX GPI is not mirrored (its owner asks to
|
|
109
|
+
be contacted before results are published); HACC region 5 is not available on Globus (HTTP 404).
|
|
110
|
+
|
|
111
|
+
Please cite SDRBench and the data provider named on each dataset card (`sdrbench cite <dataset>`), and
|
|
112
|
+
report the `sdrbench` version you used: a release always reads the same data.
|
|
113
|
+
|
|
114
|
+
## Keeping the mirror in sync
|
|
115
|
+
|
|
116
|
+
Globus remains the source of truth. `.github/workflows/release.yml` runs weekly:
|
|
117
|
+
|
|
118
|
+
1. `sync/sync.py check` HEADs every archive linked from the SDRBench page and compares size, ETag and
|
|
119
|
+
Last-Modified with `sync/state.json`; archives that appear on, or disappear from, the page open an issue.
|
|
120
|
+
2. Changed archives are re-mirrored in parallel jobs (`sync.py mirror`). Archives are streamed: each file is
|
|
121
|
+
unpacked, hashed and committed to `sdrbench/<dataset>` in batches and then deleted, so a runner only needs
|
|
122
|
+
room for the largest single file (~17 GB); the job frees disk like the SZ3 CI. Files removed from an
|
|
123
|
+
archive are removed from the repo, but only if the whole archive arrived and at most 25% of a variant
|
|
124
|
+
goes (`--allow-delete` overrides). The same command works on any machine:
|
|
125
|
+
`HF_TOKEN=... python sync/sync.py mirror --only <dataset>`.
|
|
126
|
+
3. `sync.py catalog` regenerates `src/sdrbench/catalog.json` (pinned to the new Hugging Face revisions),
|
|
127
|
+
`sync.py verify` checks dtypes, element counts and dimension order against the data, `sync.py cards`
|
|
128
|
+
refreshes the dataset cards and file tables, and a new patch version is tagged and published to PyPI.
|
|
129
|
+
|
|
130
|
+
### Adding a dataset
|
|
131
|
+
|
|
132
|
+
1. Add an entry to `sync/datasets.json` (see its `_comment`): title, provider, science description, and per
|
|
133
|
+
variant the Hugging Face `folder`, Globus `archive`, optional `select`/`exclude`, and `rules` mapping file
|
|
134
|
+
globs to dtype, C-order shape and `var`/`step` name templates.
|
|
135
|
+
2. `python sync/sync.py mirror --only <dataset> --dry --workdir /tmp/w` downloads and hashes without uploading
|
|
136
|
+
(writes `/tmp/w/state.dry.json`); fix rules until `sync.py catalog` builds without errors (it refuses files
|
|
137
|
+
without a rule or with a size that does not match the shape).
|
|
138
|
+
3. Mirror for real, then `sync.py catalog`, `sync.py verify --only <dataset>`, `sync.py cards --only <dataset>`,
|
|
139
|
+
and `python sync/selftest.py --only <dataset> --workdir /big/tmp` (every field through the package).
|
|
140
|
+
|
|
141
|
+
## Development
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
pip install -e ".[test,sz3]"
|
|
145
|
+
pytest # offline tests
|
|
146
|
+
pytest -m online # end-to-end against Hugging Face and Globus
|
|
147
|
+
python sync/sync.py verify # data-level check of every dtype/shape (network)
|
|
148
|
+
python sync/selftest.py --workdir /big/tmp # every field of every dataset through the package
|
|
149
|
+
```
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.24", "hatch-vcs>=0.4"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "sdrbench"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Download and load SDRBench scientific datasets from Hugging Face or the original Globus archives"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "BSD-3-Clause"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "SDRBench team" }]
|
|
14
|
+
keywords = ["sdrbench", "scientific data", "lossy compression", "benchmark", "hpc", "hugging face"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Intended Audience :: Science/Research",
|
|
18
|
+
"Topic :: Scientific/Engineering",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
]
|
|
21
|
+
dependencies = ["numpy>=1.21", "huggingface_hub>=0.23", "filelock>=3.8"]
|
|
22
|
+
|
|
23
|
+
[project.optional-dependencies]
|
|
24
|
+
sz3 = ["pysz>=1.1"]
|
|
25
|
+
test = ["pytest>=7"]
|
|
26
|
+
|
|
27
|
+
[project.urls]
|
|
28
|
+
Homepage = "https://sdrbench.github.io/"
|
|
29
|
+
Source = "https://github.com/szcompressor/sdrbench"
|
|
30
|
+
"Hugging Face" = "https://huggingface.co/sdrbench"
|
|
31
|
+
|
|
32
|
+
[project.scripts]
|
|
33
|
+
sdrbench = "sdrbench.cli:main"
|
|
34
|
+
|
|
35
|
+
[tool.hatch.version]
|
|
36
|
+
source = "vcs"
|
|
37
|
+
fallback-version = "0.0.0"
|
|
38
|
+
|
|
39
|
+
[tool.hatch.build.hooks.vcs]
|
|
40
|
+
version-file = "src/sdrbench/_version.py"
|
|
41
|
+
|
|
42
|
+
[tool.hatch.build.targets.wheel]
|
|
43
|
+
packages = ["src/sdrbench"]
|
|
44
|
+
|
|
45
|
+
[tool.hatch.build.targets.sdist]
|
|
46
|
+
include = ["src/sdrbench", "tests", "README.md", "LICENSE"]
|
|
47
|
+
|
|
48
|
+
[tool.pytest.ini_options]
|
|
49
|
+
testpaths = ["tests"]
|
|
50
|
+
markers = ["online: needs network access to Hugging Face and Globus"]
|
|
51
|
+
addopts = "-m 'not online'"
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""SDRBench scientific datasets as numpy arrays.
|
|
2
|
+
|
|
3
|
+
>>> import sdrbench
|
|
4
|
+
>>> sdrbench.list() # datasets; sdrbench.list(variants=True) for all variants
|
|
5
|
+
>>> nyx = sdrbench.dataset("nyx")
|
|
6
|
+
>>> nyx.fields
|
|
7
|
+
>>> t = nyx["temperature"] # memmap, shape (512, 512, 512), C order
|
|
8
|
+
>>> p = sdrbench.dataset("hurricane-isabel", "P").series("P") # 48 time steps
|
|
9
|
+
|
|
10
|
+
Files come from the Hugging Face mirror (https://huggingface.co/sdrbench), pinned to the
|
|
11
|
+
revision this release was built from; if Hugging Face cannot deliver, from the original
|
|
12
|
+
SDRBench archives on Globus. Local copies can be used via ``root=`` or ``$SDRBENCH_DATA``.
|
|
13
|
+
"""
|
|
14
|
+
from ._core import Dataset, Field, Series, cache_dir, catalog, dataset, list, load
|
|
15
|
+
|
|
16
|
+
try:
|
|
17
|
+
from ._version import __version__
|
|
18
|
+
except ImportError: # source checkout without a build
|
|
19
|
+
__version__ = "0.0.0"
|
|
20
|
+
|
|
21
|
+
# `list` is deliberately not exported by `from sdrbench import *` (it would shadow the builtin);
|
|
22
|
+
# use sdrbench.list().
|
|
23
|
+
__all__ = ["Dataset", "Field", "Series", "cache_dir", "catalog", "dataset", "load"]
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Download and unpack SDRBench archives.
|
|
2
|
+
|
|
3
|
+
The Hugging Face mirror is built with exactly this code, so a file fetched
|
|
4
|
+
from Globus ends up at the same relative path, with the same bytes, as the
|
|
5
|
+
file stored on Hugging Face.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import hashlib
|
|
10
|
+
import http.client
|
|
11
|
+
import os
|
|
12
|
+
import shutil
|
|
13
|
+
import tarfile
|
|
14
|
+
import time
|
|
15
|
+
import urllib.error
|
|
16
|
+
import urllib.request
|
|
17
|
+
import zipfile
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Optional, Tuple
|
|
20
|
+
|
|
21
|
+
CHUNK = 1 << 22
|
|
22
|
+
JUNK = ("._", ".DS_Store", "__MACOSX")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _is_junk(name: str) -> bool:
|
|
26
|
+
return any(part.startswith(JUNK) for part in Path(name).parts)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def sha256_file(path, chunk: int = 1 << 24) -> str:
|
|
30
|
+
h = hashlib.sha256()
|
|
31
|
+
with open(path, "rb") as f:
|
|
32
|
+
while b := f.read(chunk):
|
|
33
|
+
h.update(b)
|
|
34
|
+
return h.hexdigest()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class _HashingReader:
|
|
38
|
+
"""File-like wrapper that hashes and counts everything read through it."""
|
|
39
|
+
|
|
40
|
+
def __init__(self, raw):
|
|
41
|
+
self.raw = raw
|
|
42
|
+
self.md5 = hashlib.md5()
|
|
43
|
+
self.nbytes = 0
|
|
44
|
+
|
|
45
|
+
def read(self, n: int = -1) -> bytes:
|
|
46
|
+
b = self.raw.read(n)
|
|
47
|
+
self.md5.update(b)
|
|
48
|
+
self.nbytes += len(b)
|
|
49
|
+
return b
|
|
50
|
+
|
|
51
|
+
def drain(self) -> None:
|
|
52
|
+
while self.read(CHUNK):
|
|
53
|
+
pass
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class _ResumingResponse:
|
|
57
|
+
"""Readable HTTP body that reconnects with a Range request when the connection drops,
|
|
58
|
+
so multi-GB downloads survive transient network failures.
|
|
59
|
+
|
|
60
|
+
The full size must be known (Globus sends GET bodies chunked, without Content-Length, so it
|
|
61
|
+
comes from HEAD); without it a dropped connection cannot be told apart from the end of the
|
|
62
|
+
file, so we refuse to download instead of risking a silently truncated archive."""
|
|
63
|
+
|
|
64
|
+
def __init__(self, url: str, retries: int = 10, timeout: float = 120):
|
|
65
|
+
self.url, self.retries, self.timeout = url, retries, timeout
|
|
66
|
+
self.pos = 0
|
|
67
|
+
self.total = self._head_length()
|
|
68
|
+
self.resp = self._connect()
|
|
69
|
+
|
|
70
|
+
def _head_length(self) -> int:
|
|
71
|
+
last = None
|
|
72
|
+
for attempt in range(self.retries + 1):
|
|
73
|
+
req = urllib.request.Request(self.url, method="HEAD", headers={"User-Agent": "sdrbench"})
|
|
74
|
+
try:
|
|
75
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as r:
|
|
76
|
+
n = r.headers.get("Content-Length")
|
|
77
|
+
if n is not None:
|
|
78
|
+
return int(n)
|
|
79
|
+
last = "no Content-Length in HEAD response"
|
|
80
|
+
except urllib.error.HTTPError as e:
|
|
81
|
+
if e.code not in (429, 502, 503, 504): # only temporary server errors are retried
|
|
82
|
+
raise
|
|
83
|
+
last = e
|
|
84
|
+
except (OSError, http.client.HTTPException, ValueError) as e:
|
|
85
|
+
last = e
|
|
86
|
+
time.sleep(min(60, 2 ** attempt))
|
|
87
|
+
raise IOError(f"{self.url}: cannot determine the archive size ({last}); refusing an unverifiable download")
|
|
88
|
+
|
|
89
|
+
def _connect(self):
|
|
90
|
+
headers = {"User-Agent": "sdrbench"}
|
|
91
|
+
if self.pos:
|
|
92
|
+
headers["Range"] = f"bytes={self.pos}-"
|
|
93
|
+
resp = urllib.request.urlopen(urllib.request.Request(self.url, headers=headers), timeout=self.timeout)
|
|
94
|
+
if self.pos:
|
|
95
|
+
rng = resp.headers.get("Content-Range", "")
|
|
96
|
+
if resp.status != 206 or not rng.startswith(f"bytes {self.pos}-"):
|
|
97
|
+
resp.close()
|
|
98
|
+
raise IOError(f"{self.url}: server did not resume at byte {self.pos} ({resp.status} {rng!r})")
|
|
99
|
+
return resp
|
|
100
|
+
|
|
101
|
+
def read(self, n: int = -1) -> bytes:
|
|
102
|
+
for attempt in range(self.retries + 1):
|
|
103
|
+
try:
|
|
104
|
+
b = self.resp.read(n)
|
|
105
|
+
if not b and n != 0 and self.pos < self.total:
|
|
106
|
+
# body ended early: a dropped connection (with chunked encoding this can look
|
|
107
|
+
# like a clean end of stream)
|
|
108
|
+
raise http.client.IncompleteRead(b"", self.total - self.pos)
|
|
109
|
+
self.pos += len(b)
|
|
110
|
+
if self.pos > self.total:
|
|
111
|
+
raise IOError(f"{self.url}: received more than the {self.total} bytes announced")
|
|
112
|
+
return b
|
|
113
|
+
except (OSError, http.client.HTTPException) as e:
|
|
114
|
+
if isinstance(e, IOError) and "announced" in str(e) or attempt == self.retries:
|
|
115
|
+
raise
|
|
116
|
+
time.sleep(min(60, 2 ** attempt))
|
|
117
|
+
try:
|
|
118
|
+
self.resp.close()
|
|
119
|
+
except Exception:
|
|
120
|
+
pass
|
|
121
|
+
try:
|
|
122
|
+
self.resp = self._connect()
|
|
123
|
+
except (OSError, http.client.HTTPException):
|
|
124
|
+
continue # try again on the next attempt
|
|
125
|
+
raise IOError(f"{self.url}: download failed at byte {self.pos} of {self.total}")
|
|
126
|
+
|
|
127
|
+
def close(self):
|
|
128
|
+
self.resp.close()
|
|
129
|
+
|
|
130
|
+
def __enter__(self):
|
|
131
|
+
return self
|
|
132
|
+
|
|
133
|
+
def __exit__(self, *exc):
|
|
134
|
+
self.close()
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _open_url(url: str):
|
|
138
|
+
if url.startswith(("http://", "https://")):
|
|
139
|
+
return _ResumingResponse(url)
|
|
140
|
+
return urllib.request.urlopen(urllib.request.Request(url, headers={"User-Agent": "sdrbench"}), timeout=120)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _safe_target(root: Path, name: str) -> Path:
|
|
144
|
+
target = (root / name).resolve()
|
|
145
|
+
if os.path.commonpath([str(root.resolve()), str(target)]) != str(root.resolve()):
|
|
146
|
+
raise ValueError(f"unsafe path in archive: {name}")
|
|
147
|
+
return target
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _extract_tar_stream(stream, root: Path) -> None:
|
|
151
|
+
with tarfile.open(fileobj=stream, mode="r|*") as tf:
|
|
152
|
+
for m in tf:
|
|
153
|
+
if _is_junk(m.name) or not (m.isfile() or m.isdir()):
|
|
154
|
+
continue
|
|
155
|
+
target = _safe_target(root, m.name)
|
|
156
|
+
if m.isdir():
|
|
157
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
158
|
+
continue
|
|
159
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
160
|
+
src = tf.extractfile(m)
|
|
161
|
+
with open(target, "wb") as out:
|
|
162
|
+
shutil.copyfileobj(src, out, CHUNK)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _extract_zip(path: Path, root: Path) -> None:
|
|
166
|
+
with zipfile.ZipFile(path) as zf:
|
|
167
|
+
for info in zf.infolist():
|
|
168
|
+
if _is_junk(info.filename) or info.is_dir():
|
|
169
|
+
continue
|
|
170
|
+
target = _safe_target(root, info.filename)
|
|
171
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
172
|
+
with zf.open(info) as src, open(target, "wb") as out:
|
|
173
|
+
shutil.copyfileobj(src, out, CHUNK)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def fetch_archive(url: str, dest, expected_md5: Optional[str] = None) -> Tuple[str, int]:
|
|
177
|
+
"""Stream ``url`` (a .tar.gz or .zip) into directory ``dest``.
|
|
178
|
+
|
|
179
|
+
If all files of the archive live under a single top-level directory, its contents are
|
|
180
|
+
placed directly in ``dest`` (the same rule the Hugging Face mirror uses). File contents
|
|
181
|
+
are never modified. The whole archive must arrive: its size is checked against the
|
|
182
|
+
server's, and its md5 against ``expected_md5`` when given.
|
|
183
|
+
Returns ``(md5 of the archive, archive size in bytes)``.
|
|
184
|
+
"""
|
|
185
|
+
import tempfile
|
|
186
|
+
|
|
187
|
+
dest = Path(dest)
|
|
188
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
189
|
+
tmp = Path(tempfile.mkdtemp(prefix=dest.name + ".", suffix=".partial", dir=dest.parent))
|
|
190
|
+
try:
|
|
191
|
+
with _open_url(url) as resp:
|
|
192
|
+
reader = _HashingReader(resp)
|
|
193
|
+
if url.endswith(".zip"):
|
|
194
|
+
zpath = tmp / "_archive.zip"
|
|
195
|
+
with open(zpath, "wb") as out:
|
|
196
|
+
shutil.copyfileobj(reader, out, CHUNK)
|
|
197
|
+
files = tmp / "files"
|
|
198
|
+
_extract_zip(zpath, files)
|
|
199
|
+
zpath.unlink()
|
|
200
|
+
else:
|
|
201
|
+
files = tmp / "files"
|
|
202
|
+
files.mkdir()
|
|
203
|
+
_extract_tar_stream(reader, files)
|
|
204
|
+
reader.drain()
|
|
205
|
+
total = getattr(resp, "total", None)
|
|
206
|
+
if total is not None and reader.nbytes != total:
|
|
207
|
+
raise IOError(f"{url}: got {reader.nbytes} of {total} bytes")
|
|
208
|
+
md5 = reader.md5.hexdigest()
|
|
209
|
+
if expected_md5 and md5 != expected_md5:
|
|
210
|
+
raise IOError(f"md5 mismatch for {url}: got {md5}, expected {expected_md5}")
|
|
211
|
+
tops = {p.relative_to(files).parts[0] for p in files.rglob("*") if p.is_file()}
|
|
212
|
+
nested = {p.relative_to(files).parts[0] for p in files.rglob("*") if p.is_file()
|
|
213
|
+
and len(p.relative_to(files).parts) > 1}
|
|
214
|
+
src = files / next(iter(tops)) if len(tops) == 1 and tops == nested else files
|
|
215
|
+
shutil.rmtree(dest, ignore_errors=True)
|
|
216
|
+
os.replace(src, dest)
|
|
217
|
+
return md5, reader.nbytes
|
|
218
|
+
finally:
|
|
219
|
+
shutil.rmtree(tmp, ignore_errors=True)
|