pyhdf5-handler 0.9__tar.gz → 0.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/PKG-INFO +11 -2
  2. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/README.md +10 -1
  3. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/Release_note.txt +5 -0
  4. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/src/hdf5_handler.py +54 -52
  5. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/tutorial/hdf5_io_tests.py +24 -3
  6. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyproject.toml +1 -1
  7. pyhdf5_handler-0.9/mycontrol.hdf5 +0 -0
  8. pyhdf5_handler-0.9/test.hdf5 +0 -0
  9. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/backups/codestyle.ini.bak +0 -0
  10. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/backups/encoding.ini.bak +0 -0
  11. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/backups/vcs.ini.bak +0 -0
  12. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/backups/workspace.ini.bak +0 -0
  13. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/codestyle.ini +0 -0
  14. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-codestyle-0.2.0.ini +0 -0
  15. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-encoding-0.2.0.ini +0 -0
  16. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-vcs-0.2.0.ini +0 -0
  17. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-workspace-0.2.0.ini +0 -0
  18. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/encoding.ini +0 -0
  19. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/vcs.ini +0 -0
  20. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/.spyproject/config/workspace.ini +0 -0
  21. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/LICENSE +0 -0
  22. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/__init__.py +0 -0
  23. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/src/__init__.py +0 -0
  24. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/src/constant.py +0 -0
  25. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/src/object_handler.py +0 -0
  26. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/tutorial/__init__.py +0 -0
  27. {pyhdf5_handler-0.9 → pyhdf5_handler-0.10}/pyhdf5_handler/tutorial/complementary_test.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyhdf5_handler
3
- Version: 0.9
3
+ Version: 0.10
4
4
  Summary: A Python library to read and write data to HDF5 format. This library is based on the h5py (https://docs.h5py.org/en/stable/index.html) and has been developed by Maxime Jay-Allemand at Hydris hydrologie (https://www.hydris-hydrologie.fr/)
5
5
  Project-URL: Homepage, https://codeberg.org/maximejay/pyhdf5_handler
6
6
  Project-URL: Issues, https://codeberg.org/maximejay/pyhdf5_handler/issues
@@ -29,7 +29,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
29
29
  * Pandas DatetimeIndex
30
30
  * Numpy array
31
31
  * Structured numpy array
32
-
32
+ * pandas DataFrame
33
33
 
34
34
  Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
35
35
 
@@ -111,6 +111,15 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
111
111
  ```
112
112
  Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
113
113
 
114
+ Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
115
+
116
+ ```python
117
+ data = pyhdf5_handler.read_hdf5file_as_dict(
118
+ "./test.hdf5", read_attrs=False
119
+ )
120
+ ```
121
+
122
+
114
123
  #### Writing a Python "object" (class) into an HDF5 file
115
124
 
116
125
  The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
@@ -12,7 +12,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
12
12
  * Pandas DatetimeIndex
13
13
  * Numpy array
14
14
  * Structured numpy array
15
-
15
+ * pandas DataFrame
16
16
 
17
17
  Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
18
18
 
@@ -94,6 +94,15 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
94
94
  ```
95
95
  Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
96
96
 
97
+ Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
98
+
99
+ ```python
100
+ data = pyhdf5_handler.read_hdf5file_as_dict(
101
+ "./test.hdf5", read_attrs=False
102
+ )
103
+ ```
104
+
105
+
97
106
  #### Writing a Python "object" (class) into an HDF5 file
98
107
 
99
108
  The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
@@ -1,3 +1,8 @@
1
+
2
+ v0.10: Fri Jul 31 17:30:46 2026 +0200
3
+ - Improvement : writting and reading a pandas dataframe and a numpy ndarray no longer save and read it in a specific subgroup.
4
+
5
+
1
6
  v0.9 : 23-07-2026
2
7
  - fix: handle dtype with panda dataframe: read/write
3
8
 
@@ -407,8 +407,7 @@ def _hdf5_handle_PandaDataFrame(
407
407
  hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
408
408
  hdf5_data = hdf5[name]
409
409
 
410
- hdf5_data = add_hdf5_sub_group(hdf5_data, subgroup="pd_DataFrame")
411
- hdf5_data = hdf5_data["pd_DataFrame"]
410
+ hdf5_data.attrs["_Pandas_DataFrame"] = 1
412
411
 
413
412
  keys = _hdf5_handle_array("columns", np.array(list(value.columns)))
414
413
  _hdf5_create_dataset(hdf5_data, keys)
@@ -418,7 +417,12 @@ def _hdf5_handle_PandaDataFrame(
418
417
 
419
418
  dtype = []
420
419
  for col in value.columns:
421
- dtype.append(value[col].dtype.name)
420
+ if pd.api.types.is_string_dtype(value[col]):
421
+ dtype_name = "str"
422
+ else:
423
+ dtype_name = value[col].dtype.name
424
+
425
+ dtype.append(dtype_name)
422
426
 
423
427
  dataset = _hdf5_handle_list("dtype", dtype)
424
428
  _hdf5_create_dataset(hdf5_data, dataset)
@@ -499,7 +503,10 @@ def _hdf5_handle_array(name: str, value: np.ndarray):
499
503
  def _hdf5_handle_ndarray(hdf5: h5py.File, name: str, value: np.ndarray):
500
504
 
501
505
  hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
502
- _dump_ndarray_to_hdf5(hdf5[name], value)
506
+ hdf5_data = hdf5[name]
507
+
508
+ hdf5_data.attrs["_numpy_ndarray"] = 1
509
+ _dump_ndarray_to_hdf5(hdf5_data, value)
503
510
 
504
511
 
505
512
  def _hdf5_create_dataset(hdf5: h5py.File, dataset: dict):
@@ -606,13 +613,12 @@ def _dump_ndarray_to_hdf5(hdf5, value):
606
613
 
607
614
  """
608
615
  # save ndarray datastructure
609
-
610
- hdf5 = add_hdf5_sub_group(hdf5, subgroup="ndarray_ds")
611
- hdf5_data = hdf5["ndarray_ds"]
616
+ # hdf5 = add_hdf5_sub_group(hdf5, subgroup="ndarray_ds")
617
+ # hdf5_data = hdf5["ndarray_ds"]
612
618
 
613
619
  for item in value.dtype.names:
614
620
 
615
- hdf5_dataset_creator(hdf5=hdf5_data, name=item, value=value[item])
621
+ hdf5_dataset_creator(hdf5=hdf5, name=item, value=value[item])
616
622
 
617
623
  index = np.array(value.dtype.descr)[:, 0]
618
624
  dtype = np.array(value.dtype.descr)[:, 1]
@@ -620,10 +626,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
620
626
  dtype = dtype.astype("O")
621
627
  data_type = h5py.string_dtype(encoding="utf-8")
622
628
 
623
- if "ndarray_dtype" in hdf5_data.keys():
624
- del hdf5_data["ndarray_dtype"]
629
+ if "ndarray_dtype" in hdf5.keys():
630
+ del hdf5["ndarray_dtype"]
625
631
 
626
- hdf5_data.create_dataset(
632
+ hdf5.create_dataset(
627
633
  "ndarray_dtype",
628
634
  shape=dtype.shape,
629
635
  dtype=data_type,
@@ -632,10 +638,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
632
638
  chunks=True,
633
639
  )
634
640
 
635
- if "ndarray_indexes" in hdf5_data.keys():
636
- del hdf5_data["ndarray_indexes"]
641
+ if "ndarray_indexes" in hdf5.keys():
642
+ del hdf5["ndarray_indexes"]
637
643
 
638
- hdf5_data.create_dataset(
644
+ hdf5.create_dataset(
639
645
  "ndarray_indexes",
640
646
  shape=index.shape,
641
647
  dtype=data_type,
@@ -662,10 +668,10 @@ def _read_pd_dataframe(hdf5):
662
668
 
663
669
  """
664
670
 
665
- if "pd_DataFrame" in list(hdf5.keys()):
666
- columns = hdf5["pd_DataFrame/columns"][:]
667
- array = hdf5["pd_DataFrame/array"][:]
668
- dtype = hdf5["pd_DataFrame/dtype"][:]
671
+ if "_Pandas_DataFrame" in list(hdf5.attrs.keys()):
672
+ columns = hdf5["columns"][:]
673
+ array = hdf5["array"][:]
674
+ dtype = hdf5["dtype"][:]
669
675
 
670
676
  newdict = {}
671
677
  for i, col in enumerate(columns):
@@ -691,39 +697,36 @@ def _read_ndarray_datastructure(hdf5):
691
697
 
692
698
  """
693
699
 
694
- if "ndarray_ds" in list(hdf5.keys()):
695
-
696
- decoded_item = list()
697
- for it in hdf5["ndarray_ds/ndarray_dtype"][:]:
698
- decoded_item.append(it.decode())
699
- list_dtypes = decoded_item
700
+ # if "ndarray_ds" in list(hdf5.keys()):
701
+ decoded_item = list()
702
+ for it in hdf5["ndarray_dtype"][:]:
703
+ decoded_item.append(it.decode())
704
+ list_dtypes = decoded_item
700
705
 
701
- decoded_item = list()
702
- for it in hdf5["ndarray_ds/ndarray_indexes"][:]:
703
- decoded_item.append(it.decode())
704
- list_indexes = decoded_item
706
+ decoded_item = list()
707
+ for it in hdf5["ndarray_indexes"][:]:
708
+ decoded_item.append(it.decode())
709
+ list_indexes = decoded_item
705
710
 
706
- len_data = len(hdf5[f"ndarray_ds/{list_indexes[0]}"][:])
711
+ len_data = len(hdf5[f"{list_indexes[0]}"][:])
707
712
 
708
- list_datatype = list()
709
- for i in range(len(list_indexes)):
710
- list_datatype.append((list_indexes[i], list_dtypes[i]))
713
+ list_datatype = list()
714
+ for i in range(len(list_indexes)):
715
+ list_datatype.append((list_indexes[i], list_dtypes[i]))
711
716
 
712
- datatype = np.dtype(list_datatype)
717
+ datatype = np.dtype(list_datatype)
713
718
 
714
- ndarray = np.zeros(len_data, dtype=datatype)
719
+ ndarray = np.zeros(len_data, dtype=datatype)
715
720
 
716
- for i in range(len(list_indexes)):
721
+ for i in range(len(list_indexes)):
717
722
 
718
- expected_type = list_dtypes[i]
723
+ expected_type = list_dtypes[i]
719
724
 
720
- values = hdf5_read_dataset(
721
- hdf5[f"ndarray_ds/{list_indexes[i]}"], expected_type
722
- )
725
+ values = hdf5_read_dataset(hdf5[f"{list_indexes[i]}"], expected_type)
723
726
 
724
- ndarray[list_indexes[i]] = values
727
+ ndarray[list_indexes[i]] = values
725
728
 
726
- return ndarray
729
+ return ndarray
727
730
 
728
731
 
729
732
  def save_dict_to_hdf5(hdf5, dictionary):
@@ -1027,14 +1030,13 @@ def read_hdf5_as_dict(hdf5, read_attrs=True, read_dataset_attrs=False):
1027
1030
 
1028
1031
  if str(type(item)).find("group") != -1:
1029
1032
 
1030
- if key == "ndarray_ds":
1031
-
1032
- # dictionary.update({key: _read_ndarray_datastructure(hdf5)})
1033
- values = _read_ndarray_datastructure(hdf5)
1033
+ if "_Pandas_DataFrame" in list(item.attrs.keys()):
1034
+ values = _read_pd_dataframe(item)
1034
1035
  dictionary.update({key: values})
1035
1036
 
1036
- elif key == "pd_DataFrame":
1037
- values = _read_pd_dataframe(hdf5)
1037
+ elif "_numpy_ndarray" in list(item.attrs.keys()):
1038
+ values = _read_ndarray_datastructure(item)
1039
+ # values = _read_ndarray_datastructure(hdf5)
1038
1040
  dictionary.update({key: values})
1039
1041
 
1040
1042
  else:
@@ -1454,13 +1456,13 @@ def get_hdf5_item(
1454
1456
 
1455
1457
  if str(type(hdf5_item)).find("group") != -1:
1456
1458
 
1457
- if item == "ndarray_ds":
1458
-
1459
- return _read_ndarray_datastructure(hdf5)
1459
+ # if item == "ndarray_ds":
1460
+ if "_numpy_ndarray" in list(hdf5_item.attrs.keys()):
1460
1461
 
1461
- elif item == "pd_DataFrame":
1462
+ return _read_ndarray_datastructure(hdf5_item)
1462
1463
 
1463
- return _read_pd_dataframe(hdf5)
1464
+ elif "_Pandas_DataFrame" in list(hdf5_item.attrs.keys()):
1465
+ return _read_pd_dataframe(hdf5_item)
1464
1466
 
1465
1467
  else:
1466
1468
 
@@ -167,8 +167,29 @@ if __name__ == "__main__":
167
167
  pyhdf5_handler.hdf5file_ls("./test.hdf5")
168
168
  pyhdf5_handler.hdf5file_ls("./test.hdf5", location="structured_array")
169
169
 
170
- data = pyhdf5_handler.read_hdf5file_as_dict(
171
- "./test.hdf5", read_attrs=False
170
+ pyhdf5_handler.save_dict_to_hdf5file(
171
+ "./panda.hdf5",
172
+ {
173
+ "mypandas": pd.DataFrame(
174
+ {
175
+ "column1": np.array(["A", "B", "C"]),
176
+ "column2": np.array([4, 5, 6]),
177
+ }
178
+ )
179
+ },
180
+ )
181
+
182
+ pd_data = pyhdf5_handler.read_hdf5file_as_dict(
183
+ "./panda.hdf5", read_attrs=False
184
+ )
185
+
186
+ pyhdf5_handler.save_dict_to_hdf5file(
187
+ "./ndarray.hdf5",
188
+ {"myndarray": people},
189
+ )
190
+
191
+ ndarray = pyhdf5_handler.read_hdf5file_as_dict(
192
+ "./ndarray.hdf5", read_attrs=False
172
193
  )
173
194
 
174
195
  pyhdf5_handler.get_hdf5file_item(
@@ -244,7 +265,7 @@ if __name__ == "__main__":
244
265
 
245
266
  pyhdf5_handler.get_hdf5file_attribute(
246
267
  path_to_hdf5="./test.hdf5",
247
- location="./structured_array/ndarray_ds",
268
+ location="./structured_array/",
248
269
  attribute="_name",
249
270
  wait_time=0,
250
271
  )
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pyhdf5_handler"
7
- version = "0.9"
7
+ version = "0.10"
8
8
  authors = [
9
9
  { name="Maxime Jay-Allemand", email="maxime.jay.allemand@hydris-hydrologie.fr" },
10
10
  ]
Binary file
Binary file
File without changes