pyhdf5-handler 0.8__tar.gz → 0.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/backups/workspace.ini.bak +1 -1
  2. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/workspace.ini +1 -1
  3. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/PKG-INFO +11 -2
  4. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/README.md +10 -1
  5. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/Release_note.txt +8 -0
  6. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/src/hdf5_handler.py +61 -51
  7. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/tutorial/hdf5_io_tests.py +54 -3
  8. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyproject.toml +1 -1
  9. pyhdf5_handler-0.8/mycontrol.hdf5 +0 -0
  10. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/backups/codestyle.ini.bak +0 -0
  11. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/backups/encoding.ini.bak +0 -0
  12. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/backups/vcs.ini.bak +0 -0
  13. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/codestyle.ini +0 -0
  14. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-codestyle-0.2.0.ini +0 -0
  15. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-encoding-0.2.0.ini +0 -0
  16. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-vcs-0.2.0.ini +0 -0
  17. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/defaults/defaults-workspace-0.2.0.ini +0 -0
  18. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/encoding.ini +0 -0
  19. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/.spyproject/config/vcs.ini +0 -0
  20. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/LICENSE +0 -0
  21. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/__init__.py +0 -0
  22. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/src/__init__.py +0 -0
  23. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/src/constant.py +0 -0
  24. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/src/object_handler.py +0 -0
  25. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/tutorial/__init__.py +0 -0
  26. {pyhdf5_handler-0.8 → pyhdf5_handler-0.10}/pyhdf5_handler/tutorial/complementary_test.py +0 -0
@@ -4,7 +4,7 @@ save_data_on_exit = True
4
4
  save_history = True
5
5
  save_non_project_files = False
6
6
  project_type = 'empty-project-type'
7
- recent_files = ['pyhdf5_handler/src/hdf5_handler.py', 'pyhdf5_handler/src/object_handler.py', 'pyhdf5_handler/tutorial/hdf5_io_tests.py', 'pyhdf5_handler/tutorial/complementary_test.py', 'test_fichier_julie.py']
7
+ recent_files = ['pyhdf5_handler/src/hdf5_handler.py', 'pyhdf5_handler/src/object_handler.py', 'pyhdf5_handler/tutorial/hdf5_io_tests.py', 'pyhdf5_handler/tutorial/complementary_test.py']
8
8
 
9
9
  [main]
10
10
  version = 0.2.0
@@ -4,7 +4,7 @@ save_data_on_exit = True
4
4
  save_history = True
5
5
  save_non_project_files = False
6
6
  project_type = 'empty-project-type'
7
- recent_files = ['pyhdf5_handler/src/hdf5_handler.py', 'pyhdf5_handler/src/object_handler.py', 'pyhdf5_handler/tutorial/hdf5_io_tests.py', 'pyhdf5_handler/tutorial/complementary_test.py', 'test_fichier_julie.py']
7
+ recent_files = ['pyhdf5_handler/src/hdf5_handler.py', 'pyhdf5_handler/src/object_handler.py', 'pyhdf5_handler/tutorial/hdf5_io_tests.py', 'pyhdf5_handler/tutorial/complementary_test.py']
8
8
 
9
9
  [main]
10
10
  version = 0.2.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyhdf5_handler
3
- Version: 0.8
3
+ Version: 0.10
4
4
  Summary: A Python library to read and write data to HDF5 format. This library is based on the h5py (https://docs.h5py.org/en/stable/index.html) and has been developed by Maxime Jay-Allemand at Hydris hydrologie (https://www.hydris-hydrologie.fr/)
5
5
  Project-URL: Homepage, https://codeberg.org/maximejay/pyhdf5_handler
6
6
  Project-URL: Issues, https://codeberg.org/maximejay/pyhdf5_handler/issues
@@ -29,7 +29,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
29
29
  * Pandas DatetimeIndex
30
30
  * Numpy array
31
31
  * Structured numpy array
32
-
32
+ * pandas DataFrame
33
33
 
34
34
  Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
35
35
 
@@ -111,6 +111,15 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
111
111
  ```
112
112
  Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
113
113
 
114
+ Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
115
+
116
+ ```python
117
+ data = pyhdf5_handler.read_hdf5file_as_dict(
118
+ "./test.hdf5", read_attrs=False
119
+ )
120
+ ```
121
+
122
+
114
123
  #### Writing a Python "object" (class) into an HDF5 file
115
124
 
116
125
  The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
@@ -12,7 +12,7 @@ Read and write to hdf5 are currently limited to the following supported Python t
12
12
  * Pandas DatetimeIndex
13
13
  * Numpy array
14
14
  * Structured numpy array
15
-
15
+ * pandas DataFrame
16
16
 
17
17
  Basically, data are stored in hdf5 file using the h5py pythonic interface to the binary HDF5 format. All data are stored in dataset using Numpy. If the data type is not supported by the HDF5 format, the data will be converted to a supported type (byte for string). For each stored dataset, an attribute, containing the type of the original data, is auto-generated. This attribute help to convert back the data to its original type when reading the HDF5.
18
18
 
@@ -94,6 +94,15 @@ pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", dictionary)
94
94
  ```
95
95
  Function argument `location` can be used to save the dictionary in a specific group. the group will be created if not exist. The function argument `replace` can be used to replace the HDF5 file, by default pyhdf5_handler will append to it.
96
96
 
97
+ Reading an hdf5 is easy. Just use the function `pyhdf5_handler.src.hdf5_handler.read_hdf5file_as_dict`:
98
+
99
+ ```python
100
+ data = pyhdf5_handler.read_hdf5file_as_dict(
101
+ "./test.hdf5", read_attrs=False
102
+ )
103
+ ```
104
+
105
+
97
106
  #### Writing a Python "object" (class) into an HDF5 file
98
107
 
99
108
  The function `pyhdf5_handler.src.hdf5_handler.save_object_to_hdf5file` lets you write any Python class-object into an hdf5 file. The hdf5 file will mimic the structure of this class object.
@@ -1,3 +1,11 @@
1
+
2
+ v0.10: Fri Jul 31 17:30:46 2026 +0200
3
+ - Improvement : writting and reading a pandas dataframe and a numpy ndarray no longer save and read it in a specific subgroup.
4
+
5
+
6
+ v0.9 : 23-07-2026
7
+ - fix: handle dtype with panda dataframe: read/write
8
+
1
9
  v0.8 : 19-06-2026
2
10
  - relase 0.8
3
11
  - Add support for pd.DataFrame()
@@ -407,8 +407,7 @@ def _hdf5_handle_PandaDataFrame(
407
407
  hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
408
408
  hdf5_data = hdf5[name]
409
409
 
410
- hdf5_data = add_hdf5_sub_group(hdf5_data, subgroup="pd_DataFrame")
411
- hdf5_data = hdf5_data["pd_DataFrame"]
410
+ hdf5_data.attrs["_Pandas_DataFrame"] = 1
412
411
 
413
412
  keys = _hdf5_handle_array("columns", np.array(list(value.columns)))
414
413
  _hdf5_create_dataset(hdf5_data, keys)
@@ -416,6 +415,18 @@ def _hdf5_handle_PandaDataFrame(
416
415
  dataset = _hdf5_handle_array("array", value.to_numpy())
417
416
  _hdf5_create_dataset(hdf5_data, dataset)
418
417
 
418
+ dtype = []
419
+ for col in value.columns:
420
+ if pd.api.types.is_string_dtype(value[col]):
421
+ dtype_name = "str"
422
+ else:
423
+ dtype_name = value[col].dtype.name
424
+
425
+ dtype.append(dtype_name)
426
+
427
+ dataset = _hdf5_handle_list("dtype", dtype)
428
+ _hdf5_create_dataset(hdf5_data, dataset)
429
+
419
430
  return
420
431
 
421
432
 
@@ -492,7 +503,10 @@ def _hdf5_handle_array(name: str, value: np.ndarray):
492
503
  def _hdf5_handle_ndarray(hdf5: h5py.File, name: str, value: np.ndarray):
493
504
 
494
505
  hdf5 = add_hdf5_sub_group(hdf5, subgroup=name)
495
- _dump_ndarray_to_hdf5(hdf5[name], value)
506
+ hdf5_data = hdf5[name]
507
+
508
+ hdf5_data.attrs["_numpy_ndarray"] = 1
509
+ _dump_ndarray_to_hdf5(hdf5_data, value)
496
510
 
497
511
 
498
512
  def _hdf5_create_dataset(hdf5: h5py.File, dataset: dict):
@@ -599,13 +613,12 @@ def _dump_ndarray_to_hdf5(hdf5, value):
599
613
 
600
614
  """
601
615
  # save ndarray datastructure
602
-
603
- hdf5 = add_hdf5_sub_group(hdf5, subgroup="ndarray_ds")
604
- hdf5_data = hdf5["ndarray_ds"]
616
+ # hdf5 = add_hdf5_sub_group(hdf5, subgroup="ndarray_ds")
617
+ # hdf5_data = hdf5["ndarray_ds"]
605
618
 
606
619
  for item in value.dtype.names:
607
620
 
608
- hdf5_dataset_creator(hdf5=hdf5_data, name=item, value=value[item])
621
+ hdf5_dataset_creator(hdf5=hdf5, name=item, value=value[item])
609
622
 
610
623
  index = np.array(value.dtype.descr)[:, 0]
611
624
  dtype = np.array(value.dtype.descr)[:, 1]
@@ -613,10 +626,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
613
626
  dtype = dtype.astype("O")
614
627
  data_type = h5py.string_dtype(encoding="utf-8")
615
628
 
616
- if "ndarray_dtype" in hdf5_data.keys():
617
- del hdf5_data["ndarray_dtype"]
629
+ if "ndarray_dtype" in hdf5.keys():
630
+ del hdf5["ndarray_dtype"]
618
631
 
619
- hdf5_data.create_dataset(
632
+ hdf5.create_dataset(
620
633
  "ndarray_dtype",
621
634
  shape=dtype.shape,
622
635
  dtype=data_type,
@@ -625,10 +638,10 @@ def _dump_ndarray_to_hdf5(hdf5, value):
625
638
  chunks=True,
626
639
  )
627
640
 
628
- if "ndarray_indexes" in hdf5_data.keys():
629
- del hdf5_data["ndarray_indexes"]
641
+ if "ndarray_indexes" in hdf5.keys():
642
+ del hdf5["ndarray_indexes"]
630
643
 
631
- hdf5_data.create_dataset(
644
+ hdf5.create_dataset(
632
645
  "ndarray_indexes",
633
646
  shape=index.shape,
634
647
  dtype=data_type,
@@ -655,13 +668,14 @@ def _read_pd_dataframe(hdf5):
655
668
 
656
669
  """
657
670
 
658
- if "pd_DataFrame" in list(hdf5.keys()):
659
- columns = hdf5["pd_DataFrame/columns"][:]
660
- array = hdf5["pd_DataFrame/array"][:]
671
+ if "_Pandas_DataFrame" in list(hdf5.attrs.keys()):
672
+ columns = hdf5["columns"][:]
673
+ array = hdf5["array"][:]
674
+ dtype = hdf5["dtype"][:]
661
675
 
662
676
  newdict = {}
663
677
  for i, col in enumerate(columns):
664
- newdict.update({col.decode(): array[:, i]})
678
+ newdict.update({col.decode(): array[:, i].astype(dtype[i])})
665
679
 
666
680
  return pd.DataFrame(newdict)
667
681
 
@@ -683,39 +697,36 @@ def _read_ndarray_datastructure(hdf5):
683
697
 
684
698
  """
685
699
 
686
- if "ndarray_ds" in list(hdf5.keys()):
700
+ # if "ndarray_ds" in list(hdf5.keys()):
701
+ decoded_item = list()
702
+ for it in hdf5["ndarray_dtype"][:]:
703
+ decoded_item.append(it.decode())
704
+ list_dtypes = decoded_item
687
705
 
688
- decoded_item = list()
689
- for it in hdf5["ndarray_ds/ndarray_dtype"][:]:
690
- decoded_item.append(it.decode())
691
- list_dtypes = decoded_item
706
+ decoded_item = list()
707
+ for it in hdf5["ndarray_indexes"][:]:
708
+ decoded_item.append(it.decode())
709
+ list_indexes = decoded_item
692
710
 
693
- decoded_item = list()
694
- for it in hdf5["ndarray_ds/ndarray_indexes"][:]:
695
- decoded_item.append(it.decode())
696
- list_indexes = decoded_item
711
+ len_data = len(hdf5[f"{list_indexes[0]}"][:])
697
712
 
698
- len_data = len(hdf5[f"ndarray_ds/{list_indexes[0]}"][:])
713
+ list_datatype = list()
714
+ for i in range(len(list_indexes)):
715
+ list_datatype.append((list_indexes[i], list_dtypes[i]))
699
716
 
700
- list_datatype = list()
701
- for i in range(len(list_indexes)):
702
- list_datatype.append((list_indexes[i], list_dtypes[i]))
717
+ datatype = np.dtype(list_datatype)
703
718
 
704
- datatype = np.dtype(list_datatype)
719
+ ndarray = np.zeros(len_data, dtype=datatype)
705
720
 
706
- ndarray = np.zeros(len_data, dtype=datatype)
721
+ for i in range(len(list_indexes)):
707
722
 
708
- for i in range(len(list_indexes)):
723
+ expected_type = list_dtypes[i]
709
724
 
710
- expected_type = list_dtypes[i]
725
+ values = hdf5_read_dataset(hdf5[f"{list_indexes[i]}"], expected_type)
711
726
 
712
- values = hdf5_read_dataset(
713
- hdf5[f"ndarray_ds/{list_indexes[i]}"], expected_type
714
- )
727
+ ndarray[list_indexes[i]] = values
715
728
 
716
- ndarray[list_indexes[i]] = values
717
-
718
- return ndarray
729
+ return ndarray
719
730
 
720
731
 
721
732
  def save_dict_to_hdf5(hdf5, dictionary):
@@ -1019,14 +1030,13 @@ def read_hdf5_as_dict(hdf5, read_attrs=True, read_dataset_attrs=False):
1019
1030
 
1020
1031
  if str(type(item)).find("group") != -1:
1021
1032
 
1022
- if key == "ndarray_ds":
1023
-
1024
- # dictionary.update({key: _read_ndarray_datastructure(hdf5)})
1025
- values = _read_ndarray_datastructure(hdf5)
1033
+ if "_Pandas_DataFrame" in list(item.attrs.keys()):
1034
+ values = _read_pd_dataframe(item)
1026
1035
  dictionary.update({key: values})
1027
1036
 
1028
- elif key == "pd_DataFrame":
1029
- values = _read_pd_dataframe(hdf5)
1037
+ elif "_numpy_ndarray" in list(item.attrs.keys()):
1038
+ values = _read_ndarray_datastructure(item)
1039
+ # values = _read_ndarray_datastructure(hdf5)
1030
1040
  dictionary.update({key: values})
1031
1041
 
1032
1042
  else:
@@ -1446,13 +1456,13 @@ def get_hdf5_item(
1446
1456
 
1447
1457
  if str(type(hdf5_item)).find("group") != -1:
1448
1458
 
1449
- if item == "ndarray_ds":
1450
-
1451
- return _read_ndarray_datastructure(hdf5)
1459
+ # if item == "ndarray_ds":
1460
+ if "_numpy_ndarray" in list(hdf5_item.attrs.keys()):
1452
1461
 
1453
- elif item == "pd_DataFrame":
1462
+ return _read_ndarray_datastructure(hdf5_item)
1454
1463
 
1455
- return _read_pd_dataframe(hdf5)
1464
+ elif "_Pandas_DataFrame" in list(hdf5_item.attrs.keys()):
1465
+ return _read_pd_dataframe(hdf5_item)
1456
1466
 
1457
1467
  else:
1458
1468
 
@@ -74,6 +74,21 @@ if __name__ == "__main__":
74
74
  {"column1": np.array([1, 2, 3]), "column2": np.array([4, 5, 6])}
75
75
  ),
76
76
  )
77
+ pyhdf5_handler.hdf5_dataset_creator(
78
+ hdf5,
79
+ "mix_dtype_panda_dataframe",
80
+ pd.DataFrame(
81
+ {
82
+ "column1": np.array(["A", "B", "C"]),
83
+ "column2": np.array([4, 5, 6]),
84
+ }
85
+ ),
86
+ )
87
+ pyhdf5_handler.hdf5_dataset_creator(
88
+ hdf5,
89
+ "empty_dataframe",
90
+ pd.DataFrame({}),
91
+ )
77
92
 
78
93
  # write a python dictionary in the hdf5 database
79
94
  dictionary = {
@@ -86,6 +101,12 @@ if __name__ == "__main__":
86
101
  "array": np.array([1, 2, 3, 4]),
87
102
  "date_range": pd.date_range(start="1/1/2018", end="1/08/2018"),
88
103
  "list_mixte": [1.0, np.datetime64("2019-09-22 17:38:30")],
104
+ "pandas_df": pd.DataFrame(
105
+ {
106
+ "column1": np.array([1, 2, 3]),
107
+ "column2": np.array([4, 5, 6]),
108
+ }
109
+ ),
89
110
  }
90
111
  }
91
112
 
@@ -130,6 +151,10 @@ if __name__ == "__main__":
130
151
  hdf5_instance=hdf5,
131
152
  location="./panda_dataframe",
132
153
  )
154
+ pyhdf5_handler.get_hdf5_item(
155
+ hdf5_instance=hdf5,
156
+ location="./mix_dtype_panda_dataframe",
157
+ )
133
158
  pyhdf5_handler.get_hdf5_item(
134
159
  hdf5_instance=hdf5,
135
160
  location="./structured_array",
@@ -142,8 +167,34 @@ if __name__ == "__main__":
142
167
  pyhdf5_handler.hdf5file_ls("./test.hdf5")
143
168
  pyhdf5_handler.hdf5file_ls("./test.hdf5", location="structured_array")
144
169
 
145
- data = pyhdf5_handler.read_hdf5file_as_dict(
146
- "./test.hdf5", read_attrs=False
170
+ pyhdf5_handler.save_dict_to_hdf5file(
171
+ "./panda.hdf5",
172
+ {
173
+ "mypandas": pd.DataFrame(
174
+ {
175
+ "column1": np.array(["A", "B", "C"]),
176
+ "column2": np.array([4, 5, 6]),
177
+ }
178
+ )
179
+ },
180
+ )
181
+
182
+ pd_data = pyhdf5_handler.read_hdf5file_as_dict(
183
+ "./panda.hdf5", read_attrs=False
184
+ )
185
+
186
+ pyhdf5_handler.save_dict_to_hdf5file(
187
+ "./ndarray.hdf5",
188
+ {"myndarray": people},
189
+ )
190
+
191
+ ndarray = pyhdf5_handler.read_hdf5file_as_dict(
192
+ "./ndarray.hdf5", read_attrs=False
193
+ )
194
+
195
+ pyhdf5_handler.get_hdf5file_item(
196
+ path_to_hdf5="./test.hdf5",
197
+ location="./mix_dtype_panda_dataframe",
147
198
  )
148
199
 
149
200
  pyhdf5_handler.save_dict_to_hdf5file("./test.hdf5", data)
@@ -214,7 +265,7 @@ if __name__ == "__main__":
214
265
 
215
266
  pyhdf5_handler.get_hdf5file_attribute(
216
267
  path_to_hdf5="./test.hdf5",
217
- location="./structured_array/ndarray_ds",
268
+ location="./structured_array/",
218
269
  attribute="_name",
219
270
  wait_time=0,
220
271
  )
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pyhdf5_handler"
7
- version = "0.8"
7
+ version = "0.10"
8
8
  authors = [
9
9
  { name="Maxime Jay-Allemand", email="maxime.jay.allemand@hydris-hydrologie.fr" },
10
10
  ]
Binary file
File without changes