pyhdf5-handler 0.9__tar.gz → 0.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/PKG-INFO +15 -3
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/README.md +13 -1
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/Release_note.txt +6 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/hdf5_handler.py +68 -53
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/tutorial/hdf5_io_tests.py +32 -3
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyproject.toml +1 -1
- pyhdf5_handler-0.9/mycontrol.hdf5 +0 -0
- pyhdf5_handler-0.9/test.hdf5 +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/codestyle.ini.bak +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/encoding.ini.bak +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/vcs.ini.bak +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/workspace.ini.bak +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/codestyle.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-codestyle-0.2.0.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-encoding-0.2.0.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-vcs-0.2.0.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-workspace-0.2.0.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/encoding.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/vcs.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/workspace.ini +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/LICENSE +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/__init__.py +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/__init__.py +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/constant.py +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/object_handler.py +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/tutorial/__init__.py +0 -0
- {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/tutorial/complementary_test.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: pyhdf5_handler
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.11
|
|
4
4
|
Summary: A Python library to read and write data to HDF5 format. This library is based on the h5py (https://docs.h5py.org/en/stable/index.html) and has been developed by Maxime Jay-Allemand at Hydris hydrologie (https://www.hydris-hydrologie.fr/)
|
|
5
5
|
Project-URL: Homepage, https://codeberg.org/maximejay/pyhdf5_handler
|
|
6
6
|
Project-URL: Issues, https://codeberg.org/maximejay/pyhdf5_handler/issues
|
|
@@ -29,7 +29,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
|
|
|
29
29
|
* Pandas DatetimeIndex
|
|
30
30
|
* Numpy array
|
|
31
31
|
* Structured numpy array
|
|
32
|
-
|
|
32
|
+
* pandas DataFrame
|
|
33
33
|
|
|
34
34
|
Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
|
|
35
35
|
|
|
@@ -111,6 +111,18 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
|
|
|
111
111
|
```
|
|
112
112
|
Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
|
|
113
113
|
|
|
114
|
+
|
|
115
|
+
#### Reading an HDF5 file as Python dictionary
|
|
116
|
+
|
|
117
|
+
Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
data = pyhdf5_handler.read_hdf5file_as_dict(
|
|
121
|
+
"./test.hdf5", read_attrs=False
|
|
122
|
+
)
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
|
|
114
126
|
#### Writing a Python "object" (class) into an HDF5 file
|
|
115
127
|
|
|
116
128
|
The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
|
|
@@ -12,7 +12,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
|
|
|
12
12
|
* Pandas DatetimeIndex
|
|
13
13
|
* Numpy array
|
|
14
14
|
* Structured numpy array
|
|
15
|
-
|
|
15
|
+
* pandas DataFrame
|
|
16
16
|
|
|
17
17
|
Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
|
|
18
18
|
|
|
@@ -94,6 +94,18 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
|
|
|
94
94
|
```
|
|
95
95
|
Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
|
|
96
96
|
|
|
97
|
+
|
|
98
|
+
#### Reading an HDF5 file as Python dictionary
|
|
99
|
+
|
|
100
|
+
Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
data = pyhdf5_handler.read_hdf5file_as_dict(
|
|
104
|
+
"./test.hdf5", read_attrs=False
|
|
105
|
+
)
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
|
|
97
109
|
#### Writing a Python "object" (class) into an HDF5 file
|
|
98
110
|
|
|
99
111
|
The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
|
|
@@ -1,3 +1,9 @@
|
|
|
1
|
+
v0.11: thur Aug 20 12:02:40 2026 +0200
|
|
2
|
+
- Improvement : writting and reading a pandas dataframe keep the original dataframe index.
|
|
3
|
+
|
|
4
|
+
v0.10: Fri Jul 31 17:30:46 2026 +0200
|
|
5
|
+
- Improvement : writting and reading a pandas dataframe and a numpy ndarray no longer save and read it in a specific subgroup.
|
|
6
|
+
|
|
1
7
|
v0.9 : 23-07-2026
|
|
2
8
|
- fix: handle dtype with panda dataframe: read/write
|
|
3
9
|
|
|
@@ -407,18 +407,25 @@ def _hdf5_handle_PandaDataFrame(
|
|
|
407
407
|
hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
|
|
408
408
|
hdf5_data = hdf5[name]
|
|
409
409
|
|
|
410
|
-
hdf5_data =
|
|
411
|
-
hdf5_data = hdf5_data["pd_DataFrame"]
|
|
410
|
+
hdf5_data.attrs["_Pandas_DataFrame"] = 1
|
|
412
411
|
|
|
413
412
|
keys = _hdf5_handle_array("columns", np.array(list(value.columns)))
|
|
414
413
|
_hdf5_create_dataset(hdf5_data, keys)
|
|
415
414
|
|
|
415
|
+
index = _hdf5_handle_array("index", np.array(list(value.index)))
|
|
416
|
+
_hdf5_create_dataset(hdf5_data, index)
|
|
417
|
+
|
|
416
418
|
dataset = _hdf5_handle_array("array", value.to_numpy())
|
|
417
419
|
_hdf5_create_dataset(hdf5_data, dataset)
|
|
418
420
|
|
|
419
421
|
dtype = []
|
|
420
422
|
for col in value.columns:
|
|
421
|
-
|
|
423
|
+
if pd.api.types.is_string_dtype(value[col]):
|
|
424
|
+
dtype_name = "str"
|
|
425
|
+
else:
|
|
426
|
+
dtype_name = value[col].dtype.name
|
|
427
|
+
|
|
428
|
+
dtype.append(dtype_name)
|
|
422
429
|
|
|
423
430
|
dataset = _hdf5_handle_list("dtype", dtype)
|
|
424
431
|
_hdf5_create_dataset(hdf5_data, dataset)
|
|
@@ -499,7 +506,10 @@ def _hdf5_handle_array(name: str, value: np.ndarray):
|
|
|
499
506
|
def _hdf5_handle_ndarray(hdf5: h5py.File, name: str, value: np.ndarray):
|
|
500
507
|
|
|
501
508
|
hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
|
|
502
|
-
|
|
509
|
+
hdf5_data = hdf5[name]
|
|
510
|
+
|
|
511
|
+
hdf5_data.attrs["_numpy_ndarray"] = 1
|
|
512
|
+
_dump_ndarray_to_hdf5(hdf5_data, value)
|
|
503
513
|
|
|
504
514
|
|
|
505
515
|
def _hdf5_create_dataset(hdf5: h5py.File, dataset: dict):
|
|
@@ -606,13 +616,12 @@ def _dump_ndarray_to_hdf5(hdf5, value):
|
|
|
606
616
|
|
|
607
617
|
"""
|
|
608
618
|
# save ndarray datastructure
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
hdf5_data = hdf5["ndarray_ds"]
|
|
619
|
+
# hdf5 = add_hdf5_sub_group(hdf5, subgroup="ndarray_ds")
|
|
620
|
+
# hdf5_data = hdf5["ndarray_ds"]
|
|
612
621
|
|
|
613
622
|
for item in value.dtype.names:
|
|
614
623
|
|
|
615
|
-
hdf5_dataset_creator(hdf5=
|
|
624
|
+
hdf5_dataset_creator(hdf5=hdf5, name=item, value=value[item])
|
|
616
625
|
|
|
617
626
|
index = np.array(value.dtype.descr)[:, 0]
|
|
618
627
|
dtype = np.array(value.dtype.descr)[:, 1]
|
|
@@ -620,10 +629,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
|
|
|
620
629
|
dtype = dtype.astype("O")
|
|
621
630
|
data_type = h5py.string_dtype(encoding="utf-8")
|
|
622
631
|
|
|
623
|
-
if "ndarray_dtype" in
|
|
624
|
-
del
|
|
632
|
+
if "ndarray_dtype" in hdf5.keys():
|
|
633
|
+
del hdf5["ndarray_dtype"]
|
|
625
634
|
|
|
626
|
-
|
|
635
|
+
hdf5.create_dataset(
|
|
627
636
|
"ndarray_dtype",
|
|
628
637
|
shape=dtype.shape,
|
|
629
638
|
dtype=data_type,
|
|
@@ -632,10 +641,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
|
|
|
632
641
|
chunks=True,
|
|
633
642
|
)
|
|
634
643
|
|
|
635
|
-
if "ndarray_indexes" in
|
|
636
|
-
del
|
|
644
|
+
if "ndarray_indexes" in hdf5.keys():
|
|
645
|
+
del hdf5["ndarray_indexes"]
|
|
637
646
|
|
|
638
|
-
|
|
647
|
+
hdf5.create_dataset(
|
|
639
648
|
"ndarray_indexes",
|
|
640
649
|
shape=index.shape,
|
|
641
650
|
dtype=data_type,
|
|
@@ -662,16 +671,26 @@ def _read_pd_dataframe(hdf5):
|
|
|
662
671
|
|
|
663
672
|
"""
|
|
664
673
|
|
|
665
|
-
if "
|
|
666
|
-
columns = hdf5["
|
|
667
|
-
|
|
668
|
-
|
|
674
|
+
if "_Pandas_DataFrame" in list(hdf5.attrs.keys()):
|
|
675
|
+
columns = hdf5["columns"][:]
|
|
676
|
+
index = hdf5["index"][:]
|
|
677
|
+
array = hdf5["array"][:]
|
|
678
|
+
dtype = hdf5["dtype"][:]
|
|
669
679
|
|
|
670
680
|
newdict = {}
|
|
671
681
|
for i, col in enumerate(columns):
|
|
672
682
|
newdict.update({col.decode(): array[:, i].astype(dtype[i])})
|
|
673
683
|
|
|
674
|
-
|
|
684
|
+
panda_df = pd.DataFrame(newdict)
|
|
685
|
+
|
|
686
|
+
if index.dtype.type is np.bytes_:
|
|
687
|
+
index = index.astype("str")
|
|
688
|
+
elif index.dtype.type is np.object_:
|
|
689
|
+
index = index.astype("str")
|
|
690
|
+
|
|
691
|
+
panda_df.index = index
|
|
692
|
+
|
|
693
|
+
return panda_df
|
|
675
694
|
|
|
676
695
|
|
|
677
696
|
def _read_ndarray_datastructure(hdf5):
|
|
@@ -691,39 +710,36 @@ def _read_ndarray_datastructure(hdf5):
|
|
|
691
710
|
|
|
692
711
|
"""
|
|
693
712
|
|
|
694
|
-
if "ndarray_ds" in list(hdf5.keys()):
|
|
713
|
+
# if "ndarray_ds" in list(hdf5.keys()):
|
|
714
|
+
decoded_item = list()
|
|
715
|
+
for it in hdf5["ndarray_dtype"][:]:
|
|
716
|
+
decoded_item.append(it.decode())
|
|
717
|
+
list_dtypes = decoded_item
|
|
695
718
|
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
719
|
+
decoded_item = list()
|
|
720
|
+
for it in hdf5["ndarray_indexes"][:]:
|
|
721
|
+
decoded_item.append(it.decode())
|
|
722
|
+
list_indexes = decoded_item
|
|
700
723
|
|
|
701
|
-
|
|
702
|
-
for it in hdf5["ndarray_ds/ndarray_indexes"][:]:
|
|
703
|
-
decoded_item.append(it.decode())
|
|
704
|
-
list_indexes = decoded_item
|
|
724
|
+
len_data = len(hdf5[f"{list_indexes[0]}"][:])
|
|
705
725
|
|
|
706
|
-
|
|
726
|
+
list_datatype = list()
|
|
727
|
+
for i in range(len(list_indexes)):
|
|
728
|
+
list_datatype.append((list_indexes[i], list_dtypes[i]))
|
|
707
729
|
|
|
708
|
-
|
|
709
|
-
for i in range(len(list_indexes)):
|
|
710
|
-
list_datatype.append((list_indexes[i], list_dtypes[i]))
|
|
730
|
+
datatype = np.dtype(list_datatype)
|
|
711
731
|
|
|
712
|
-
|
|
732
|
+
ndarray = np.zeros(len_data, dtype=datatype)
|
|
713
733
|
|
|
714
|
-
|
|
734
|
+
for i in range(len(list_indexes)):
|
|
715
735
|
|
|
716
|
-
|
|
736
|
+
expected_type = list_dtypes[i]
|
|
717
737
|
|
|
718
|
-
|
|
738
|
+
values = hdf5_read_dataset(hdf5[f"{list_indexes[i]}"], expected_type)
|
|
719
739
|
|
|
720
|
-
|
|
721
|
-
hdf5[f"ndarray_ds/{list_indexes[i]}"], expected_type
|
|
722
|
-
)
|
|
740
|
+
ndarray[list_indexes[i]] = values
|
|
723
741
|
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
return ndarray
|
|
742
|
+
return ndarray
|
|
727
743
|
|
|
728
744
|
|
|
729
745
|
def save_dict_to_hdf5(hdf5, dictionary):
|
|
@@ -1027,14 +1043,13 @@ def read_hdf5_as_dict(hdf5, read_attrs=True, read_dataset_attrs=False):
|
|
|
1027
1043
|
|
|
1028
1044
|
if str(type(item)).find("group") != -1:
|
|
1029
1045
|
|
|
1030
|
-
if
|
|
1031
|
-
|
|
1032
|
-
# dictionary.update({key: _read_ndarray_datastructure(hdf5)})
|
|
1033
|
-
values = _read_ndarray_datastructure(hdf5)
|
|
1046
|
+
if "_Pandas_DataFrame" in list(item.attrs.keys()):
|
|
1047
|
+
values = _read_pd_dataframe(item)
|
|
1034
1048
|
dictionary.update({key: values})
|
|
1035
1049
|
|
|
1036
|
-
elif
|
|
1037
|
-
values =
|
|
1050
|
+
elif "_numpy_ndarray" in list(item.attrs.keys()):
|
|
1051
|
+
values = _read_ndarray_datastructure(item)
|
|
1052
|
+
# values = _read_ndarray_datastructure(hdf5)
|
|
1038
1053
|
dictionary.update({key: values})
|
|
1039
1054
|
|
|
1040
1055
|
else:
|
|
@@ -1454,13 +1469,13 @@ def get_hdf5_item(
|
|
|
1454
1469
|
|
|
1455
1470
|
if str(type(hdf5_item)).find("group") != -1:
|
|
1456
1471
|
|
|
1457
|
-
if item == "ndarray_ds":
|
|
1458
|
-
|
|
1459
|
-
return _read_ndarray_datastructure(hdf5)
|
|
1472
|
+
# if item == "ndarray_ds":
|
|
1473
|
+
if "_numpy_ndarray" in list(hdf5_item.attrs.keys()):
|
|
1460
1474
|
|
|
1461
|
-
|
|
1475
|
+
return _read_ndarray_datastructure(hdf5_item)
|
|
1462
1476
|
|
|
1463
|
-
|
|
1477
|
+
elif "_Pandas_DataFrame" in list(hdf5_item.attrs.keys()):
|
|
1478
|
+
return _read_pd_dataframe(hdf5_item)
|
|
1464
1479
|
|
|
1465
1480
|
else:
|
|
1466
1481
|
|
|
@@ -167,8 +167,30 @@ if __name__ == "__main__":
|
|
|
167
167
|
pyhdf5_handler.hdf5file_ls("./test.hdf5")
|
|
168
168
|
pyhdf5_handler.hdf5file_ls("./test.hdf5", location="structured_array")
|
|
169
169
|
|
|
170
|
-
|
|
171
|
-
"./
|
|
170
|
+
pyhdf5_handler.save_dict_to_hdf5file(
|
|
171
|
+
"./panda.hdf5",
|
|
172
|
+
{
|
|
173
|
+
"mypandas": pd.DataFrame(
|
|
174
|
+
{
|
|
175
|
+
"column1": np.array(["A", "B", "C"]),
|
|
176
|
+
"column2": np.array([4, 5, 6]),
|
|
177
|
+
},
|
|
178
|
+
index=["index1", "index2", "index3"],
|
|
179
|
+
)
|
|
180
|
+
},
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
pd_data = pyhdf5_handler.read_hdf5file_as_dict(
|
|
184
|
+
"./panda.hdf5", read_attrs=False
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
pyhdf5_handler.save_dict_to_hdf5file(
|
|
188
|
+
"./ndarray.hdf5",
|
|
189
|
+
{"myndarray": people},
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
ndarray = pyhdf5_handler.read_hdf5file_as_dict(
|
|
193
|
+
"./ndarray.hdf5", read_attrs=False
|
|
172
194
|
)
|
|
173
195
|
|
|
174
196
|
pyhdf5_handler.get_hdf5file_item(
|
|
@@ -221,6 +243,13 @@ if __name__ == "__main__":
|
|
|
221
243
|
search_attrs=False,
|
|
222
244
|
)
|
|
223
245
|
|
|
246
|
+
pyhdf5_handler.get_hdf5file_item(
|
|
247
|
+
path_to_hdf5="./test.hdf5",
|
|
248
|
+
location="./",
|
|
249
|
+
item="mix_dtype_panda_dataframe",
|
|
250
|
+
search_attrs=False,
|
|
251
|
+
)
|
|
252
|
+
|
|
224
253
|
pyhdf5_handler.get_hdf5file_item(
|
|
225
254
|
path_to_hdf5="./test.hdf5",
|
|
226
255
|
location="./",
|
|
@@ -244,7 +273,7 @@ if __name__ == "__main__":
|
|
|
244
273
|
|
|
245
274
|
pyhdf5_handler.get_hdf5file_attribute(
|
|
246
275
|
path_to_hdf5="./test.hdf5",
|
|
247
|
-
location="./structured_array/
|
|
276
|
+
location="./structured_array/",
|
|
248
277
|
attribute="_name",
|
|
249
278
|
wait_time=0,
|
|
250
279
|
)
|
|
Binary file
|
pyhdf5_handler-0.9/test.hdf5
DELETED
|
Binary file
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-codestyle-0.2.0.ini
RENAMED
|
File without changes
|
{pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-encoding-0.2.0.ini
RENAMED
|
File without changes
|
{pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-vcs-0.2.0.ini
RENAMED
|
File without changes
|
{pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-workspace-0.2.0.ini
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|