pyhdf5-handler 0.9__tar.gz → 0.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/PKG-INFO +15 -3
  2. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/README.md +13 -1
  3. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/Release_note.txt +6 -0
  4. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/hdf5_handler.py +68 -53
  5. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/tutorial/hdf5_io_tests.py +32 -3
  6. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyproject.toml +1 -1
  7. pyhdf5_handler-0.9/mycontrol.hdf5 +0 -0
  8. pyhdf5_handler-0.9/test.hdf5 +0 -0
  9. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/codestyle.ini.bak +0 -0
  10. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/encoding.ini.bak +0 -0
  11. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/vcs.ini.bak +0 -0
  12. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/backups/workspace.ini.bak +0 -0
  13. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/codestyle.ini +0 -0
  14. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-codestyle-0.2.0.ini +0 -0
  15. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-encoding-0.2.0.ini +0 -0
  16. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-vcs-0.2.0.ini +0 -0
  17. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/defaults/defaults-workspace-0.2.0.ini +0 -0
  18. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/encoding.ini +0 -0
  19. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/vcs.ini +0 -0
  20. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/.spyproject/config/workspace.ini +0 -0
  21. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/LICENSE +0 -0
  22. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/__init__.py +0 -0
  23. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/__init__.py +0 -0
  24. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/constant.py +0 -0
  25. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/src/object_handler.py +0 -0
  26. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/tutorial/__init__.py +0 -0
  27. {pyhdf5_handler-0.9 → pyhdf5_handler-0.11}/pyhdf5_handler/tutorial/complementary_test.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: pyhdf5_handler
3
- Version: 0.9
3
+ Version: 0.11
4
4
  Summary: A Python library to read and write data to HDF5 format. This library is based on the h5py (https://docs.h5py.org/en/stable/index.html) and has been developed by Maxime Jay-Allemand at Hydris hydrologie (https://www.hydris-hydrologie.fr/)
5
5
  Project-URL: Homepage, https://codeberg.org/maximejay/pyhdf5_handler
6
6
  Project-URL: Issues, https://codeberg.org/maximejay/pyhdf5_handler/issues
@@ -29,7 +29,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
29
29
  * Pandas DatetimeIndex
30
30
  * Numpy array
31
31
  * Structured numpy array
32
-
32
+ * pandas DataFrame
33
33
 
34
34
  Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
35
35
 
@@ -111,6 +111,18 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
111
111
  ```
112
112
  Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
113
113
 
114
+
115
+ #### Reading an HDF5 file as Python dictionary
116
+
117
+ Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
118
+
119
+ ```python
120
+ data = pyhdf5_handler.read_hdf5file_as_dict(
121
+ "./test.hdf5", read_attrs=False
122
+ )
123
+ ```
124
+
125
+
114
126
  #### Writing a Python "object" (class) into an HDF5 file
115
127
 
116
128
  The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
@@ -12,7 +12,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
12
12
  * Pandas DatetimeIndex
13
13
  * Numpy array
14
14
  * Structured numpy array
15
-
15
+ * pandas DataFrame
16
16
 
17
17
  Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
18
18
 
@@ -94,6 +94,18 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
94
94
  ```
95
95
  Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
96
96
 
97
+
98
+ #### Reading an HDF5 file as Python dictionary
99
+
100
+ Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
101
+
102
+ ```python
103
+ data = pyhdf5_handler.read_hdf5file_as_dict(
104
+ "./test.hdf5", read_attrs=False
105
+ )
106
+ ```
107
+
108
+
97
109
  #### Writing a Python "object" (class) into an HDF5 file
98
110
 
99
111
  The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
@@ -1,3 +1,9 @@
1
+ v0.11: thur Aug 20 12:02:40 2026 +0200
2
+ - Improvement : writting and reading a pandas dataframe keep the original dataframe index.
3
+
4
+ v0.10: Fri Jul 31 17:30:46 2026 +0200
5
+ - Improvement : writting and reading a pandas dataframe and a numpy ndarray no longer save and read it in a specific subgroup.
6
+
1
7
  v0.9 : 23-07-2026
2
8
  - fix: handle dtype with panda dataframe: read/write
3
9
 
@@ -407,18 +407,25 @@ def _hdf5_handle_PandaDataFrame(
407
407
  hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
408
408
  hdf5_data = hdf5[name]
409
409
 
410
- hdf5_data = add_hdf5_sub_group(hdf5_data, subgroup="pd_DataFrame")
411
- hdf5_data = hdf5_data["pd_DataFrame"]
410
+ hdf5_data.attrs["_Pandas_DataFrame"] = 1
412
411
 
413
412
  keys = _hdf5_handle_array("columns", np.array(list(value.columns)))
414
413
  _hdf5_create_dataset(hdf5_data, keys)
415
414
 
415
+ index = _hdf5_handle_array("index", np.array(list(value.index)))
416
+ _hdf5_create_dataset(hdf5_data, index)
417
+
416
418
  dataset = _hdf5_handle_array("array", value.to_numpy())
417
419
  _hdf5_create_dataset(hdf5_data, dataset)
418
420
 
419
421
  dtype = []
420
422
  for col in value.columns:
421
- dtype.append(value[col].dtype.name)
423
+ if pd.api.types.is_string_dtype(value[col]):
424
+ dtype_name = "str"
425
+ else:
426
+ dtype_name = value[col].dtype.name
427
+
428
+ dtype.append(dtype_name)
422
429
 
423
430
  dataset = _hdf5_handle_list("dtype", dtype)
424
431
  _hdf5_create_dataset(hdf5_data, dataset)
@@ -499,7 +506,10 @@ def _hdf5_handle_array(name: str, value: np.ndarray):
499
506
  def _hdf5_handle_ndarray(hdf5: h5py.File, name: str, value: np.ndarray):
500
507
 
501
508
  hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
502
- _dump_ndarray_to_hdf5(hdf5[name], value)
509
+ hdf5_data = hdf5[name]
510
+
511
+ hdf5_data.attrs["_numpy_ndarray"] = 1
512
+ _dump_ndarray_to_hdf5(hdf5_data, value)
503
513
 
504
514
 
505
515
  def _hdf5_create_dataset(hdf5: h5py.File, dataset: dict):
@@ -606,13 +616,12 @@ def _dump_ndarray_to_hdf5(hdf5, value):
606
616
 
607
617
  """
608
618
  # save ndarray datastructure
609
-
610
- hdf5 = add_hdf5_sub_group(hdf5, subgroup="ndarray_ds")
611
- hdf5_data = hdf5["ndarray_ds"]
619
+ # hdf5 = add_hdf5_sub_group(hdf5, subgroup="ndarray_ds")
620
+ # hdf5_data = hdf5["ndarray_ds"]
612
621
 
613
622
  for item in value.dtype.names:
614
623
 
615
- hdf5_dataset_creator(hdf5=hdf5_data, name=item, value=value[item])
624
+ hdf5_dataset_creator(hdf5=hdf5, name=item, value=value[item])
616
625
 
617
626
  index = np.array(value.dtype.descr)[:, 0]
618
627
  dtype = np.array(value.dtype.descr)[:, 1]
@@ -620,10 +629,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
620
629
  dtype = dtype.astype("O")
621
630
  data_type = h5py.string_dtype(encoding="utf-8")
622
631
 
623
- if "ndarray_dtype" in hdf5_data.keys():
624
- del hdf5_data["ndarray_dtype"]
632
+ if "ndarray_dtype" in hdf5.keys():
633
+ del hdf5["ndarray_dtype"]
625
634
 
626
- hdf5_data.create_dataset(
635
+ hdf5.create_dataset(
627
636
  "ndarray_dtype",
628
637
  shape=dtype.shape,
629
638
  dtype=data_type,
@@ -632,10 +641,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
632
641
  chunks=True,
633
642
  )
634
643
 
635
- if "ndarray_indexes" in hdf5_data.keys():
636
- del hdf5_data["ndarray_indexes"]
644
+ if "ndarray_indexes" in hdf5.keys():
645
+ del hdf5["ndarray_indexes"]
637
646
 
638
- hdf5_data.create_dataset(
647
+ hdf5.create_dataset(
639
648
  "ndarray_indexes",
640
649
  shape=index.shape,
641
650
  dtype=data_type,
@@ -662,16 +671,26 @@ def _read_pd_dataframe(hdf5):
662
671
 
663
672
  """
664
673
 
665
- if "pd_DataFrame" in list(hdf5.keys()):
666
- columns = hdf5["pd_DataFrame/columns"][:]
667
- array = hdf5["pd_DataFrame/array"][:]
668
- dtype = hdf5["pd_DataFrame/dtype"][:]
674
+ if "_Pandas_DataFrame" in list(hdf5.attrs.keys()):
675
+ columns = hdf5["columns"][:]
676
+ index = hdf5["index"][:]
677
+ array = hdf5["array"][:]
678
+ dtype = hdf5["dtype"][:]
669
679
 
670
680
  newdict = {}
671
681
  for i, col in enumerate(columns):
672
682
  newdict.update({col.decode(): array[:, i].astype(dtype[i])})
673
683
 
674
- return pd.DataFrame(newdict)
684
+ panda_df = pd.DataFrame(newdict)
685
+
686
+ if index.dtype.type is np.bytes_:
687
+ index = index.astype("str")
688
+ elif index.dtype.type is np.object_:
689
+ index = index.astype("str")
690
+
691
+ panda_df.index = index
692
+
693
+ return panda_df
675
694
 
676
695
 
677
696
  def _read_ndarray_datastructure(hdf5):
@@ -691,39 +710,36 @@ def _read_ndarray_datastructure(hdf5):
691
710
 
692
711
  """
693
712
 
694
- if "ndarray_ds" in list(hdf5.keys()):
713
+ # if "ndarray_ds" in list(hdf5.keys()):
714
+ decoded_item = list()
715
+ for it in hdf5["ndarray_dtype"][:]:
716
+ decoded_item.append(it.decode())
717
+ list_dtypes = decoded_item
695
718
 
696
- decoded_item = list()
697
- for it in hdf5["ndarray_ds/ndarray_dtype"][:]:
698
- decoded_item.append(it.decode())
699
- list_dtypes = decoded_item
719
+ decoded_item = list()
720
+ for it in hdf5["ndarray_indexes"][:]:
721
+ decoded_item.append(it.decode())
722
+ list_indexes = decoded_item
700
723
 
701
- decoded_item = list()
702
- for it in hdf5["ndarray_ds/ndarray_indexes"][:]:
703
- decoded_item.append(it.decode())
704
- list_indexes = decoded_item
724
+ len_data = len(hdf5[f"{list_indexes[0]}"][:])
705
725
 
706
- len_data = len(hdf5[f"ndarray_ds/{list_indexes[0]}"][:])
726
+ list_datatype = list()
727
+ for i in range(len(list_indexes)):
728
+ list_datatype.append((list_indexes[i], list_dtypes[i]))
707
729
 
708
- list_datatype = list()
709
- for i in range(len(list_indexes)):
710
- list_datatype.append((list_indexes[i], list_dtypes[i]))
730
+ datatype = np.dtype(list_datatype)
711
731
 
712
- datatype = np.dtype(list_datatype)
732
+ ndarray = np.zeros(len_data, dtype=datatype)
713
733
 
714
- ndarray = np.zeros(len_data, dtype=datatype)
734
+ for i in range(len(list_indexes)):
715
735
 
716
- for i in range(len(list_indexes)):
736
+ expected_type = list_dtypes[i]
717
737
 
718
- expected_type = list_dtypes[i]
738
+ values = hdf5_read_dataset(hdf5[f"{list_indexes[i]}"], expected_type)
719
739
 
720
- values = hdf5_read_dataset(
721
- hdf5[f"ndarray_ds/{list_indexes[i]}"], expected_type
722
- )
740
+ ndarray[list_indexes[i]] = values
723
741
 
724
- ndarray[list_indexes[i]] = values
725
-
726
- return ndarray
742
+ return ndarray
727
743
 
728
744
 
729
745
  def save_dict_to_hdf5(hdf5, dictionary):
@@ -1027,14 +1043,13 @@ def read_hdf5_as_dict(hdf5, read_attrs=True, read_dataset_attrs=False):
1027
1043
 
1028
1044
  if str(type(item)).find("group") != -1:
1029
1045
 
1030
- if key == "ndarray_ds":
1031
-
1032
- # dictionary.update({key: _read_ndarray_datastructure(hdf5)})
1033
- values = _read_ndarray_datastructure(hdf5)
1046
+ if "_Pandas_DataFrame" in list(item.attrs.keys()):
1047
+ values = _read_pd_dataframe(item)
1034
1048
  dictionary.update({key: values})
1035
1049
 
1036
- elif key == "pd_DataFrame":
1037
- values = _read_pd_dataframe(hdf5)
1050
+ elif "_numpy_ndarray" in list(item.attrs.keys()):
1051
+ values = _read_ndarray_datastructure(item)
1052
+ # values = _read_ndarray_datastructure(hdf5)
1038
1053
  dictionary.update({key: values})
1039
1054
 
1040
1055
  else:
@@ -1454,13 +1469,13 @@ def get_hdf5_item(
1454
1469
 
1455
1470
  if str(type(hdf5_item)).find("group") != -1:
1456
1471
 
1457
- if item == "ndarray_ds":
1458
-
1459
- return _read_ndarray_datastructure(hdf5)
1472
+ # if item == "ndarray_ds":
1473
+ if "_numpy_ndarray" in list(hdf5_item.attrs.keys()):
1460
1474
 
1461
- elif item == "pd_DataFrame":
1475
+ return _read_ndarray_datastructure(hdf5_item)
1462
1476
 
1463
- return _read_pd_dataframe(hdf5)
1477
+ elif "_Pandas_DataFrame" in list(hdf5_item.attrs.keys()):
1478
+ return _read_pd_dataframe(hdf5_item)
1464
1479
 
1465
1480
  else:
1466
1481
 
@@ -167,8 +167,30 @@ if __name__ == "__main__":
167
167
  pyhdf5_handler.hdf5file_ls("./test.hdf5")
168
168
  pyhdf5_handler.hdf5file_ls("./test.hdf5", location="structured_array")
169
169
 
170
- data = pyhdf5_handler.read_hdf5file_as_dict(
171
- "./test.hdf5", read_attrs=False
170
+ pyhdf5_handler.save_dict_to_hdf5file(
171
+ "./panda.hdf5",
172
+ {
173
+ "mypandas": pd.DataFrame(
174
+ {
175
+ "column1": np.array(["A", "B", "C"]),
176
+ "column2": np.array([4, 5, 6]),
177
+ },
178
+ index=["index1", "index2", "index3"],
179
+ )
180
+ },
181
+ )
182
+
183
+ pd_data = pyhdf5_handler.read_hdf5file_as_dict(
184
+ "./panda.hdf5", read_attrs=False
185
+ )
186
+
187
+ pyhdf5_handler.save_dict_to_hdf5file(
188
+ "./ndarray.hdf5",
189
+ {"myndarray": people},
190
+ )
191
+
192
+ ndarray = pyhdf5_handler.read_hdf5file_as_dict(
193
+ "./ndarray.hdf5", read_attrs=False
172
194
  )
173
195
 
174
196
  pyhdf5_handler.get_hdf5file_item(
@@ -221,6 +243,13 @@ if __name__ == "__main__":
221
243
  search_attrs=False,
222
244
  )
223
245
 
246
+ pyhdf5_handler.get_hdf5file_item(
247
+ path_to_hdf5="./test.hdf5",
248
+ location="./",
249
+ item="mix_dtype_panda_dataframe",
250
+ search_attrs=False,
251
+ )
252
+
224
253
  pyhdf5_handler.get_hdf5file_item(
225
254
  path_to_hdf5="./test.hdf5",
226
255
  location="./",
@@ -244,7 +273,7 @@ if __name__ == "__main__":
244
273
 
245
274
  pyhdf5_handler.get_hdf5file_attribute(
246
275
  path_to_hdf5="./test.hdf5",
247
- location="./structured_array/ndarray_ds",
276
+ location="./structured_array/",
248
277
  attribute="_name",
249
278
  wait_time=0,
250
279
  )
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pyhdf5_handler"
7
- version = "0.9"
7
+ version = "0.11"
8
8
  authors = [
9
9
  { name="Maxime Jay-Allemand", email="maxime.jay.allemand@hydris-hydrologie.fr" },
10
10
  ]
Binary file
Binary file
File without changes