PyPI - disdrodb - Versions diffs - 0.1.5__py3-none-any.whl → 0.2.1__py3-none-any.whl - Mend

disdrodb 0.1.5py3-none-any.whl → 0.2.1py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (125) hide show

disdrodb/__init__.py +1 -5
disdrodb/_version.py +2 -2
disdrodb/accessor/methods.py +22 -4
disdrodb/api/checks.py +10 -0
disdrodb/api/io.py +20 -18
disdrodb/api/path.py +42 -77
disdrodb/api/search.py +89 -23
disdrodb/cli/disdrodb_create_summary.py +1 -1
disdrodb/cli/disdrodb_run_l0.py +1 -1
disdrodb/cli/disdrodb_run_l0a.py +1 -1
disdrodb/cli/disdrodb_run_l0b.py +1 -1
disdrodb/cli/disdrodb_run_l0c.py +1 -1
disdrodb/cli/disdrodb_run_l1.py +1 -1
disdrodb/cli/disdrodb_run_l2e.py +1 -1
disdrodb/cli/disdrodb_run_l2m.py +1 -1
disdrodb/configs.py +30 -83
disdrodb/constants.py +4 -3
disdrodb/data_transfer/download_data.py +4 -2
disdrodb/docs.py +2 -2
disdrodb/etc/products/L1/1MIN.yaml +13 -0
disdrodb/etc/products/L1/LPM/1MIN.yaml +13 -0
disdrodb/etc/products/L1/LPM_V0/1MIN.yaml +13 -0
disdrodb/etc/products/L1/PARSIVEL/1MIN.yaml +13 -0
disdrodb/etc/products/L1/PARSIVEL2/1MIN.yaml +13 -0
disdrodb/etc/products/L1/PWS100/1MIN.yaml +13 -0
disdrodb/etc/products/L1/RD80/1MIN.yaml +13 -0
disdrodb/etc/products/L1/SWS250/1MIN.yaml +13 -0
disdrodb/etc/products/L1/global.yaml +6 -0
disdrodb/etc/products/L2E/10MIN.yaml +1 -12
disdrodb/etc/products/L2E/global.yaml +1 -1
disdrodb/etc/products/L2M/MODELS/NGAMMA_GS_R_MAE.yaml +6 -0
disdrodb/etc/products/L2M/global.yaml +1 -1
disdrodb/issue/checks.py +2 -2
disdrodb/l0/check_configs.py +1 -1
disdrodb/l0/configs/LPM/l0a_encodings.yml +0 -1
disdrodb/l0/configs/LPM/l0b_cf_attrs.yml +0 -4
disdrodb/l0/configs/LPM/l0b_encodings.yml +9 -9
disdrodb/l0/configs/LPM/raw_data_format.yml +11 -11
disdrodb/l0/configs/LPM_V0/bins_diameter.yml +103 -0
disdrodb/l0/configs/LPM_V0/bins_velocity.yml +103 -0
disdrodb/l0/configs/LPM_V0/l0a_encodings.yml +45 -0
disdrodb/l0/configs/LPM_V0/l0b_cf_attrs.yml +180 -0
disdrodb/l0/configs/LPM_V0/l0b_encodings.yml +410 -0
disdrodb/l0/configs/LPM_V0/raw_data_format.yml +474 -0
disdrodb/l0/configs/PARSIVEL/l0b_encodings.yml +1 -1
disdrodb/l0/configs/PARSIVEL/raw_data_format.yml +8 -8
disdrodb/l0/configs/PARSIVEL2/raw_data_format.yml +9 -9
disdrodb/l0/l0_reader.py +2 -2
disdrodb/l0/l0a_processing.py +6 -2
disdrodb/l0/l0b_processing.py +26 -19
disdrodb/l0/l0c_processing.py +17 -3
disdrodb/l0/manuals/LPM_V0.pdf +0 -0
disdrodb/l0/readers/LPM/ITALY/GID_LPM.py +15 -7
disdrodb/l0/readers/LPM/ITALY/GID_LPM_PI.py +279 -0
disdrodb/l0/readers/LPM/ITALY/GID_LPM_T.py +276 -0
disdrodb/l0/readers/LPM/ITALY/GID_LPM_W.py +2 -2
disdrodb/l0/readers/LPM/NETHERLANDS/DELFT_RWANDA_LPM_NC.py +103 -0
disdrodb/l0/readers/LPM/NORWAY/HAUKELISETER_LPM.py +216 -0
disdrodb/l0/readers/LPM/NORWAY/NMBU_LPM.py +208 -0
disdrodb/l0/readers/LPM/UK/WITHWORTH_LPM.py +219 -0
disdrodb/l0/readers/LPM/USA/CHARLESTON.py +229 -0
disdrodb/l0/readers/{LPM → LPM_V0}/BELGIUM/ULIEGE.py +33 -49
disdrodb/l0/readers/LPM_V0/ITALY/GID_LPM_V0.py +240 -0
disdrodb/l0/readers/PARSIVEL/BASQUECOUNTRY/EUSKALMET_OTT.py +227 -0
disdrodb/l0/readers/{PARSIVEL2 → PARSIVEL}/NASA/LPVEX.py +16 -28
disdrodb/l0/readers/PARSIVEL/{GPM → NASA}/MC3E.py +1 -1
disdrodb/l0/readers/PARSIVEL/NCAR/VORTEX2_2010_UF.py +3 -3
disdrodb/l0/readers/PARSIVEL2/BASQUECOUNTRY/EUSKALMET_OTT2.py +232 -0
disdrodb/l0/readers/PARSIVEL2/DENMARK/EROSION_raw.py +1 -1
disdrodb/l0/readers/PARSIVEL2/JAPAN/PRECIP.py +155 -0
disdrodb/l0/readers/PARSIVEL2/MPI/BCO_PARSIVEL2.py +14 -7
disdrodb/l0/readers/PARSIVEL2/MPI/BOWTIE.py +8 -3
disdrodb/l0/readers/PARSIVEL2/NASA/APU.py +28 -5
disdrodb/l0/readers/PARSIVEL2/NCAR/RELAMPAGO_PARSIVEL2.py +1 -1
disdrodb/l0/readers/PARSIVEL2/{GPM/GCPEX.py → NORWAY/UIB.py} +54 -29
disdrodb/l0/readers/PARSIVEL2/PHILIPPINES/{PANGASA.py → PAGASA.py} +6 -3
disdrodb/l0/readers/PARSIVEL2/SPAIN/GRANADA.py +1 -1
disdrodb/l0/readers/PARSIVEL2/SWEDEN/SMHI.py +189 -0
disdrodb/l0/readers/{PARSIVEL/GPM/PIERS.py → PARSIVEL2/USA/CSU.py} +62 -29
disdrodb/l0/readers/PARSIVEL2/USA/{C3WE.py → CW3E.py} +51 -24
disdrodb/l0/readers/{PARSIVEL/GPM/IFLOODS.py → RD80/BRAZIL/ATTO_RD80.py} +50 -34
disdrodb/l0/readers/{SW250 → SWS250}/BELGIUM/KMI.py +1 -1
disdrodb/l1/beard_model.py +45 -1
disdrodb/l1/fall_velocity.py +1 -6
disdrodb/l1/filters.py +2 -0
disdrodb/l1/processing.py +6 -5
disdrodb/l1/resampling.py +101 -38
disdrodb/l2/empirical_dsd.py +12 -8
disdrodb/l2/processing.py +4 -3
disdrodb/metadata/search.py +3 -4
disdrodb/routines/l0.py +4 -4
disdrodb/routines/l1.py +173 -60
disdrodb/routines/l2.py +121 -269
disdrodb/routines/options.py +347 -0
disdrodb/routines/wrappers.py +9 -1
disdrodb/scattering/axis_ratio.py +3 -0
disdrodb/scattering/routines.py +1 -1
disdrodb/summary/routines.py +765 -724
disdrodb/utils/archiving.py +51 -44
disdrodb/utils/attrs.py +1 -1
disdrodb/utils/compression.py +4 -2
disdrodb/utils/dask.py +35 -15
disdrodb/utils/dict.py +33 -0
disdrodb/utils/encoding.py +1 -1
disdrodb/utils/manipulations.py +7 -1
disdrodb/utils/routines.py +9 -8
disdrodb/utils/time.py +9 -1
disdrodb/viz/__init__.py +0 -13
disdrodb/viz/plots.py +209 -0
{disdrodb-0.1.5.dist-info → disdrodb-0.2.1.dist-info}/METADATA +1 -1
{disdrodb-0.1.5.dist-info → disdrodb-0.2.1.dist-info}/RECORD +124 -95
disdrodb/l0/readers/PARSIVEL/GPM/LPVEX.py +0 -85
/disdrodb/etc/products/L2M/{GAMMA_GS_ND_MAE.yaml → MODELS/GAMMA_GS_ND_MAE.yaml} +0 -0
/disdrodb/etc/products/L2M/{GAMMA_ML.yaml → MODELS/GAMMA_ML.yaml} +0 -0
/disdrodb/etc/products/L2M/{LOGNORMAL_GS_LOG_ND_MAE.yaml → MODELS/LOGNORMAL_GS_LOG_ND_MAE.yaml} +0 -0
/disdrodb/etc/products/L2M/{LOGNORMAL_GS_ND_MAE.yaml → MODELS/LOGNORMAL_GS_ND_MAE.yaml} +0 -0
/disdrodb/etc/products/L2M/{LOGNORMAL_ML.yaml → MODELS/LOGNORMAL_ML.yaml} +0 -0
/disdrodb/etc/products/L2M/{NGAMMA_GS_LOG_ND_MAE.yaml → MODELS/NGAMMA_GS_LOG_ND_MAE.yaml} +0 -0
/disdrodb/etc/products/L2M/{NGAMMA_GS_ND_MAE.yaml → MODELS/NGAMMA_GS_ND_MAE.yaml} +0 -0
/disdrodb/etc/products/L2M/{NGAMMA_GS_Z_MAE.yaml → MODELS/NGAMMA_GS_Z_MAE.yaml} +0 -0
/disdrodb/l0/readers/PARSIVEL2/{GPM → NASA}/NSSTC.py +0 -0
{disdrodb-0.1.5.dist-info → disdrodb-0.2.1.dist-info}/WHEEL +0 -0
{disdrodb-0.1.5.dist-info → disdrodb-0.2.1.dist-info}/entry_points.txt +0 -0
{disdrodb-0.1.5.dist-info → disdrodb-0.2.1.dist-info}/licenses/LICENSE +0 -0
{disdrodb-0.1.5.dist-info → disdrodb-0.2.1.dist-info}/top_level.txt +0 -0

disdrodb/l0/l0b_processing.py CHANGED Viewed

@@ -91,7 +91,7 @@ def format_string_array(string: str, n_values: int) -> np.array:
         e.g. : format_string_array("2,44,22,33", 4) will return [ 2. 44. 22. 33.]
-    If empty string ("") --> Return an arrays of zeros
+    If empty string ("") or "" --> Return an arrays of zeros
     If the list length is not n_values -> Return an arrays of np.nan
     The function strip potential delimiters at start and end before splitting.
@@ -108,31 +108,38 @@ def format_string_array(string: str, n_values: int) -> np.array:
     np.array
         array of float
     """
-    split_str = infer_split_str(string)
-    values = np.array(string.strip(split_str).split(split_str))
-    # -------------------------------------------------------------------------.
-    ## Assumptions !!!
-    # If empty list --> Assume no precipitation recorded. Return an arrays of zeros
-    if len(values) == 0:
+    # Check for empty string or "0" case
+    # - Assume no precipitation recorded. Return an arrays of zeros
+    if string in {"", "0"}:
         values = np.zeros(n_values)
         return values
-    # -------------------------------------------------------------------------.
+    # Check for NaN case
+    # - Assume no data available. Return an arrays of NaN
+    if string == "NaN":
+        values = np.zeros(n_values) * np.nan
+        return values
+    # Retrieve list of values
+    split_str = infer_split_str(string)
+    values = np.array(string.strip(split_str).split(split_str))
     # If the length is not as expected --> Assume data corruption
     # --> Return an array with nan
     if len(values) != n_values:
         values = np.zeros(n_values) * np.nan
-    else:
-        # Ensure string type
-        values = values.astype("str")
-        # Replace '' with 0
-        values = replace_empty_strings_with_zeros(values)
-        # Replace "-9.999" with 0
-        values = np.char.replace(values, "-9.999", "0")
-        # Cast values to float type
-        # --> Note: the disk encoding is specified in the l0b_encodings.yml
-        values = values.astype(float)
+        return values
+    # Otherwise sanitize the list of value
+    # Ensure string type
+    values = values.astype("str")
+    # Replace '' with 0
+    values = replace_empty_strings_with_zeros(values)
+    # Replace "-9.999" with 0
+    values = np.char.replace(values, "-9.999", "0")
+    # Cast values to float type
+    # --> Note: the disk encoding is specified in the l0b_encodings.yml
+    values = values.astype(float)
     return values

disdrodb/l0/l0c_processing.py CHANGED Viewed

@@ -117,7 +117,12 @@ def split_dataset_by_sampling_intervals(
     # If sample_interval is a dataset variable, use it to define dictionary of datasets
     if "sample_interval" in ds:
-        return {int(interval): ds.isel(time=ds["sample_interval"] == interval) for interval in measurement_intervals}
+        dict_ds = {}
+        for interval in measurement_intervals:
+            ds_subset = ds.isel(time=ds["sample_interval"] == interval)
+            if ds_subset.sizes["time"] > 2:
+                dict_ds[int(interval)] = ds_subset
+        return dict_ds
     # ---------------------------------------------------------------------------------------.
     # Otherwise exploit difference between timesteps to identify change point
@@ -460,9 +465,8 @@ def regularize_timesteps(ds, sample_interval, robust=False, add_quality_flag=Tru
         # if last_time == last_time_expected and qc_flag[-1] == flag_next_missing:
         #     qc_flag[-1] = 0
-        # Assign time quality flag coordinate
+        # Add time quality flag variable
         ds["time_qc"] = xr.DataArray(qc_flag, dims="time")
-        ds = ds.set_coords("time_qc")
         # Add CF attributes for time_qc
         ds["time_qc"].attrs = {
@@ -674,6 +678,16 @@ def create_l0c_datasets(
         log_info(logger=logger, msg=f"No data between {start_time} and {end_time}.", verbose=verbose)
         return {}
+    # If 1 or 2 timesteps per time block, return empty dictionary
+    n_timesteps = len(ds["time"])
+    if n_timesteps < 3:
+        log_info(
+            logger=logger,
+            msg=f"Only {n_timesteps} timesteps between {start_time} and {end_time}.",
+            verbose=verbose,
+        )
+        return {}
     # ---------------------------------------------------------------------------------------.
     # If sample interval is a dataset variable, drop timesteps with unexpected measurement intervals !
     if "sample_interval" in ds:

disdrodb/l0/manuals/LPM_V0.pdf ADDED Viewed

Binary file

disdrodb/l0/readers/LPM/ITALY/GID_LPM.py CHANGED Viewed

@@ -31,7 +31,7 @@ def reader(
     """Reader."""
     ##------------------------------------------------------------------------.
     #### - Define raw data headers
-    column_names = ["TO_BE_SPLITTED"]
+    column_names = ["TO_PARSE"]
     ##------------------------------------------------------------------------.
     #### Define reader options
@@ -79,14 +79,22 @@ def reader(
     ##------------------------------------------------------------------------.
     #### Adapt the dataframe to adhere to DISDRODB L0 standards
-    # Count number of delimiters to identify valid rows
-    df = df[df["TO_BE_SPLITTED"].str.count(";") == 519]
+    # Raise error if empty file
+    if len(df) == 0:
+        raise ValueError(f"{filepath} is empty.")
+    # Select only rows with expected number of delimiters
+    df = df[df["TO_PARSE"].str.count(";").isin([519, 520])]
+    # Check there are still valid rows
+    if len(df) == 0:
+        raise ValueError(f"No valid rows in {filepath}.")
     # Split by ; delimiter (before raw drop number)
-    df = df["TO_BE_SPLITTED"].str.split(";", expand=True, n=79)
+    df = df["TO_PARSE"].str.split(";", expand=True, n=79)
     # Assign column names
-    column_names = [
+    names = [
         "start_identifier",
         "device_address",
         "sensor_serial_number",
@@ -168,10 +176,10 @@ def reader(
         "number_particles_class_9_internal_data",
         "raw_drop_number",
     ]
-    df.columns = column_names
+    df.columns = names
     # Remove checksum from raw_drop_number
-    df["raw_drop_number"] = df["raw_drop_number"].str.rsplit(";", n=1, expand=True)[0]
+    df["raw_drop_number"] = df["raw_drop_number"].str.strip(";").str.rsplit(";", n=1, expand=True)[0]
     # Define datetime "time" column
     df["time"] = df["sensor_date"] + "-" + df["sensor_time"]

disdrodb/l0/readers/LPM/ITALY/GID_LPM_PI.py ADDED Viewed

@@ -0,0 +1,279 @@
+# -----------------------------------------------------------------------------.
+# Copyright (c) 2021-2023 DISDRODB developers
+#
+# This program is free software: you can redistribute it and/or modify
+# it under the terms of the GNU General Public License as published by
+# the Free Software Foundation, either version 3 of the License, or
+# (at your option) any later version.
+#
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+# GNU General Public License for more details.
+#
+# You should have received a copy of the GNU General Public License
+# along with this program.  If not, see <http://www.gnu.org/licenses/>.
+# -----------------------------------------------------------------------------.
+"""DISDRODB reader for GID LPM sensor TC-PI with incorrect reported time."""
+import pandas as pd
+from disdrodb.l0.l0_reader import is_documented_by, reader_generic_docstring
+from disdrodb.l0.l0a_processing import read_raw_text_file
+from disdrodb.utils.logger import log_error
+def read_txt_file(file, filename, logger):
+    """Parse for TC-PI LPM file."""
+    #### - Define raw data headers
+    column_names = ["TO_PARSE"]
+    ##------------------------------------------------------------------------.
+    #### Define reader options
+    # - For more info: https://pandas.pydata.org/docs/reference/api/pandas.read_csv.html
+    reader_kwargs = {}
+    # - Define delimiter
+    reader_kwargs["delimiter"] = "\\n"
+    # - Avoid first column to become df index !!!
+    reader_kwargs["index_col"] = False
+    # Since column names are expected to be passed explicitly, header is set to None
+    reader_kwargs["header"] = None
+    # - Number of rows to be skipped at the beginning of the file
+    reader_kwargs["skiprows"] = 1
+    # - Define behaviour when encountering bad lines
+    reader_kwargs["on_bad_lines"] = "skip"
+    # - Define reader engine
+    #   - C engine is faster
+    #   - Python engine is more feature-complete
+    reader_kwargs["engine"] = "python"
+    # - Define on-the-fly decompression of on-disk data
+    #   - Available: gzip, bz2, zip
+    reader_kwargs["compression"] = "infer"
+    # - Strings to recognize as NA/NaN and replace with standard NA flags
+    #   - Already included: '#N/A', '#N/A N/A', '#NA', '-1.#IND', '-1.#QNAN',
+    #                       '-NaN', '-nan', '1.#IND', '1.#QNAN', '<NA>', 'N/A',
+    #                       'NA', 'NULL', 'NaN', 'n/a', 'nan', 'null'
+    reader_kwargs["na_values"] = ["na", "", "error"]
+    ##------------------------------------------------------------------------.
+    #### Read the data
+    df = read_raw_text_file(
+        filepath=file,
+        column_names=column_names,
+        reader_kwargs=reader_kwargs,
+        logger=logger,
+    )
+    ##------------------------------------------------------------------------.
+    #### Adapt the dataframe to adhere to DISDRODB L0 standards
+    # Raise error if empty file
+    if len(df) == 0:
+        raise ValueError(f"{filename} is empty.")
+    # Select only rows with expected number of delimiters
+    df = df[df["TO_PARSE"].str.count(" ") == 526]
+    # Check there are still valid rows
+    if len(df) == 0:
+        raise ValueError(f"No valid rows in {filename}.")
+    # Split by ; delimiter (before raw drop number)
+    df = df["TO_PARSE"].str.split(" ", expand=True, n=82)
+    # Assign column names
+    names = [
+        "date",
+        "time",
+        "unknown",
+        "start_identifier",
+        "device_address",
+        "sensor_serial_number",
+        "sensor_date",
+        "sensor_time",
+        "weather_code_synop_4677_5min",
+        "weather_code_synop_4680_5min",
+        "weather_code_metar_4678_5min",
+        "precipitation_rate_5min",
+        "weather_code_synop_4677",
+        "weather_code_synop_4680",
+        "weather_code_metar_4678",
+        "precipitation_rate",
+        "rainfall_rate",
+        "snowfall_rate",
+        "precipitation_accumulated",
+        "mor_visibility",
+        "reflectivity",
+        "quality_index",
+        "max_hail_diameter",
+        "laser_status",
+        "static_signal_status",
+        "laser_temperature_analog_status",
+        "laser_temperature_digital_status",
+        "laser_current_analog_status",
+        "laser_current_digital_status",
+        "sensor_voltage_supply_status",
+        "current_heating_pane_transmitter_head_status",
+        "current_heating_pane_receiver_head_status",
+        "temperature_sensor_status",
+        "current_heating_voltage_supply_status",
+        "current_heating_house_status",
+        "current_heating_heads_status",
+        "current_heating_carriers_status",
+        "control_output_laser_power_status",
+        "reserved_status",
+        "temperature_interior",
+        "laser_temperature",
+        "laser_current_average",
+        "control_voltage",
+        "optical_control_voltage_output",
+        "sensor_voltage_supply",
+        "current_heating_pane_transmitter_head",
+        "current_heating_pane_receiver_head",
+        "temperature_ambient",
+        "current_heating_voltage_supply",
+        "current_heating_house",
+        "current_heating_heads",
+        "current_heating_carriers",
+        "number_particles",
+        "number_particles_internal_data",
+        "number_particles_min_speed",
+        "number_particles_min_speed_internal_data",
+        "number_particles_max_speed",
+        "number_particles_max_speed_internal_data",
+        "number_particles_min_diameter",
+        "number_particles_min_diameter_internal_data",
+        "number_particles_no_hydrometeor",
+        "number_particles_no_hydrometeor_internal_data",
+        "number_particles_unknown_classification",
+        "number_particles_unknown_classification_internal_data",
+        "number_particles_class_1",
+        "number_particles_class_1_internal_data",
+        "number_particles_class_2",
+        "number_particles_class_2_internal_data",
+        "number_particles_class_3",
+        "number_particles_class_3_internal_data",
+        "number_particles_class_4",
+        "number_particles_class_4_internal_data",
+        "number_particles_class_5",
+        "number_particles_class_5_internal_data",
+        "number_particles_class_6",
+        "number_particles_class_6_internal_data",
+        "number_particles_class_7",
+        "number_particles_class_7_internal_data",
+        "number_particles_class_8",
+        "number_particles_class_8_internal_data",
+        "number_particles_class_9",
+        "number_particles_class_9_internal_data",
+        "TO_BE_FURTHER_PROCESSED",
+    ]
+    df.columns = names
+    # Define datetime "time" column
+    df["time"] = df["date"] + " " + df["time"]
+    df["time"] = pd.to_datetime(df["time"], format="%Y-%m-%d %H:%M:%S", errors="coerce")
+    # Drop row if start_identifier different than 00
+    df = df[df["start_identifier"].astype(str) == "00"]
+    # Extract the last variables remained in raw_drop_number
+    df_parsed = df["TO_BE_FURTHER_PROCESSED"].str.rsplit(" ", n=5, expand=True)
+    df_parsed.columns = [
+        "raw_drop_number",
+        "air_temperature",
+        "relative_humidity",
+        "wind_speed",
+        "wind_direction",
+        "checksum",
+    ]
+    # Assign columns to the original dataframe
+    df[df_parsed.columns] = df_parsed
+    # Drop rows with invalid raw_drop_number
+    # --> 440 value # 22x20
+    df = df[df["raw_drop_number"].astype(str).str.len() == 1759]
+    # Drop columns not agreeing with DISDRODB L0 standards
+    columns_to_drop = [
+        "start_identifier",
+        "device_address",
+        "sensor_serial_number",
+        "sensor_date",
+        "sensor_time",
+        "date",
+        "unknown",
+        "TO_BE_FURTHER_PROCESSED",
+        "air_temperature",
+        "relative_humidity",
+        "wind_speed",
+        "wind_direction",
+        "checksum",
+    ]
+    df = df.drop(columns=columns_to_drop)
+    return df
+@is_documented_by(reader_generic_docstring)
+def reader(
+    filepath,
+    logger=None,
+):
+    """Reader."""
+    import zipfile
+    ##------------------------------------------------------------------------.
+    # filename = os.path.basename(filepath)
+    # return read_txt_file(file=filepath, filename=filename, logger=logger)
+    # ---------------------------------------------------------------------.
+    #### Iterate over all files (aka timesteps) in the daily zip archive
+    # - Each file contain a single timestep !
+    # list_df = []
+    # with tempfile.TemporaryDirectory() as temp_dir:
+    #     # Extract all files
+    #     unzip_file_on_terminal(filepath, temp_dir)
+    #     # Walk through extracted files
+    #     for root, _, files in os.walk(temp_dir):
+    #         for filename in sorted(files):
+    #             if filename.endswith(".txt"):
+    #                 full_path = os.path.join(root, filename)
+    #                 try:
+    #                     df = read_txt_file(file=full_path, filename=filename, logger=logger)
+    #                     if df is not None:
+    #                         list_df.append(df)
+    #                 except Exception as e:
+    #                     msg = f"An error occurred while reading {filename}: {e}"
+    #                     log_error(logger=logger, msg=msg, verbose=True)
+    list_df = []
+    with zipfile.ZipFile(filepath, "r") as zip_ref:
+        filenames = sorted(zip_ref.namelist())
+        for filename in filenames:
+            if filename.endswith(".txt"):
+                # Open file
+                with zip_ref.open(filename) as file:
+                    try:
+                        df = read_txt_file(file=file, filename=filename, logger=logger)
+                        if df is not None:
+                            list_df.append(df)
+                    except Exception as e:
+                        msg = f"An error occurred while reading {filename}. The error is: {e}"
+                        log_error(logger=logger, msg=msg, verbose=True)
+    # Check the zip file contains at least some non.empty files
+    if len(list_df) == 0:
+        raise ValueError(f"{filepath} contains only empty files!")
+    # Concatenate all dataframes into a single one
+    df = pd.concat(list_df)
+    # ---------------------------------------------------------------------.
+    return df

disdrodb 0.1.5__py3-none-any.whl → 0.2.1__py3-none-any.whl

disdrodb 0.1.5py3-none-any.whl → 0.2.1py3-none-any.whl