pyepwmorph 3.2.0__tar.gz → 3.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/.gitignore +4 -0
  2. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/PKG-INFO +24 -1
  3. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/README.md +23 -0
  4. pyepwmorph-3.3.0/pyepwmorph/data/amy_diffuse_table.parquet +0 -0
  5. pyepwmorph-3.3.0/pyepwmorph/tools/amy.py +867 -0
  6. pyepwmorph-3.3.0/pyepwmorph/tools/amy_meteoswiss.py +123 -0
  7. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyproject.toml +1 -1
  8. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/LICENSE +0 -0
  9. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/__init__.py +0 -0
  10. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/data/__init__.py +0 -0
  11. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/data/ch2025_monthly.parquet +0 -0
  12. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/data/ch2025_stations.parquet +0 -0
  13. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/models/__init__.py +0 -0
  14. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/models/access.py +0 -0
  15. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/models/assemble.py +0 -0
  16. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/models/ch2025.py +0 -0
  17. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/models/coordinate.py +0 -0
  18. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/models/custom.py +0 -0
  19. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/morph/__init__.py +0 -0
  20. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/morph/procedures.py +0 -0
  21. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/__init__.py +0 -0
  22. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/cache.py +0 -0
  23. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/configuration.py +0 -0
  24. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/io.py +0 -0
  25. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/psychrometrics.py +0 -0
  26. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/solar.py +0 -0
  27. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/utilities.py +0 -0
  28. {pyepwmorph-3.2.0 → pyepwmorph-3.3.0}/pyepwmorph/tools/workflow.py +0 -0
@@ -159,3 +159,7 @@ cython_debug/
159
159
  # and can be added to the global gitignore or merged into this file. For a more nuclear
160
160
  # option (not recommended) you can uncomment the following to ignore the entire idea folder.
161
161
  .idea/
162
+
163
+ # Example inputs and outputs of examples/run_amy_meteoswiss.py
164
+ examples/amy_data/
165
+ examples/amy_result/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyepwmorph
3
- Version: 3.2.0
3
+ Version: 3.3.0
4
4
  Summary: A python package to enable simple and easy gathering of climate model data and morphing of EPW files
5
5
  Project-URL: Homepage, https://github.com/justinfmccarty/pyepwmorph
6
6
  Project-URL: Issues, https://github.com/justinfmccarty/pyepwmorph/issues
@@ -152,6 +152,29 @@ The signal is the change from the 1991-2020 reference climate to the chosen warm
152
152
 
153
153
  CH2025 data: MeteoSwiss & ETH Zurich (2025), Climate CH2025 - Daily Datasets, CC-BY 4.0, https://doi.org/10.18751/climate/scenarios/ch2025/data/1.0/
154
154
 
155
+ ### Measured data (actual meteorological year)
156
+
157
+ Turn one real year of station measurements, hourly or 10-minute, into an EPW
158
+ for calibration. The table needs fixed column names and hour-ending
159
+ timestamps; `pyepwmorph.tools.amy.describe_columns()` lists them.
160
+
161
+ ```python
162
+ from pyepwmorph.tools import amy, amy_meteoswiss
163
+
164
+ table = amy_meteoswiss.read_meteoswiss_ogd("ogd-smn_sma_t_historical_2020-2029.csv") # UTC, hour-ending
165
+ location = amy_meteoswiss.meteoswiss_location("ogd-smn_meta_stations.csv", "SMA")
166
+ report = amy.station_table_to_epw(
167
+ table, location, 2025, "zurich_fluntern_2025.epw",
168
+ source_name="MeteoSwiss Zurich-Fluntern", attribution=amy_meteoswiss.ATTRIBUTION,
169
+ )
170
+ print(report.to_text())
171
+ ```
172
+
173
+ Other networks only need a table with the documented columns and
174
+ `timestamp_label` / `table_utc_offset` set to match their clock. Direct and
175
+ diffuse irradiance and sky cover are derived when the station does not measure
176
+ them (the report and the EPW header say how).
177
+
155
178
  ## Climate scenarios
156
179
 
157
180
  | Scenario | SSP | Description | Expected warming |
@@ -113,6 +113,29 @@ The signal is the change from the 1991-2020 reference climate to the chosen warm
113
113
 
114
114
  CH2025 data: MeteoSwiss & ETH Zurich (2025), Climate CH2025 - Daily Datasets, CC-BY 4.0, https://doi.org/10.18751/climate/scenarios/ch2025/data/1.0/
115
115
 
116
+ ### Measured data (actual meteorological year)
117
+
118
+ Turn one real year of station measurements, hourly or 10-minute, into an EPW
119
+ for calibration. The table needs fixed column names and hour-ending
120
+ timestamps; `pyepwmorph.tools.amy.describe_columns()` lists them.
121
+
122
+ ```python
123
+ from pyepwmorph.tools import amy, amy_meteoswiss
124
+
125
+ table = amy_meteoswiss.read_meteoswiss_ogd("ogd-smn_sma_t_historical_2020-2029.csv") # UTC, hour-ending
126
+ location = amy_meteoswiss.meteoswiss_location("ogd-smn_meta_stations.csv", "SMA")
127
+ report = amy.station_table_to_epw(
128
+ table, location, 2025, "zurich_fluntern_2025.epw",
129
+ source_name="MeteoSwiss Zurich-Fluntern", attribution=amy_meteoswiss.ATTRIBUTION,
130
+ )
131
+ print(report.to_text())
132
+ ```
133
+
134
+ Other networks only need a table with the documented columns and
135
+ `timestamp_label` / `table_utc_offset` set to match their clock. Direct and
136
+ diffuse irradiance and sky cover are derived when the station does not measure
137
+ them (the report and the EPW header say how).
138
+
116
139
  ## Climate scenarios
117
140
 
118
141
  | Scenario | SSP | Description | Expected warming |
@@ -0,0 +1,867 @@
1
+ """Build an EPW from measured weather station data (an actual meteorological year).
2
+
3
+ A typical meteorological year stitches months from many years. An *actual*
4
+ meteorological year (AMY) is one real calendar year of measurements, which is
5
+ what building energy model calibration needs. This module turns a table of
6
+ station measurements, hourly or finer, into an EPW file.
7
+
8
+ Input table
9
+ -----------
10
+ ``table`` is a ``pandas.DataFrame`` with a tz-naive ``DatetimeIndex`` and the
11
+ columns below. Column names are fixed (see :data:`COLUMN_SPEC` and
12
+ :func:`describe_columns`); an adapter such as
13
+ :mod:`pyepwmorph.tools.amy_meteoswiss` renames a network's own columns.
14
+ Unknown columns are ignored and columns that are entirely empty are dropped.
15
+
16
+ ========================= =========== =====================================================
17
+ column unit meaning
18
+ ========================= =========== =====================================================
19
+ ``temp_C`` degC dry-bulb air temperature (required)
20
+ ``dewpoint_C`` degC dew point (this or ``rh_pct`` required)
21
+ ``rh_pct`` % relative humidity (this or ``dewpoint_C`` required)
22
+ ``pressure_Pa`` Pa station pressure (default: standard atmosphere)
23
+ ``ghi_Wm2`` W/m2 global horizontal irradiance, interval mean (required)
24
+ ``dhi_Wm2``, ``dni_Wm2`` W/m2 measured diffuse / direct normal, if the site has them
25
+ ``lw_down_Wm2`` W/m2 downwelling longwave at the surface, interval mean
26
+ ``sunshine_min`` minutes sunshine duration within the interval (WMO, >=120 W/m2)
27
+ ``wind_speed_ms`` m/s wind speed, interval mean (required)
28
+ ``wind_dir_deg`` degrees direction the wind blows from, 0-360, 0 or 360 = north
29
+ ``precip_mm`` mm precipitation depth within the interval
30
+ ``snow_depth_cm`` cm snow depth at the end of the interval
31
+ ``total_sky_cover_tenths`` tenths 0-10 observed sky cover (derived when absent)
32
+ ``opaque_sky_cover_tenths`` tenths 0-10 observed opaque sky cover (set equal to total when absent)
33
+ ``visibility_km`` km visibility
34
+ ``ceiling_height_m`` m cloud ceiling height
35
+ ========================= =========== =====================================================
36
+
37
+ Datetime convention
38
+ -------------------
39
+ Every value describes an *interval*, not an instant. Say which end of the
40
+ interval the index marks with ``timestamp_label`` (``"end"`` or ``"start"``)
41
+ and which clock it uses with ``table_utc_offset`` (hours east of UTC, 0 for UTC
42
+ timestamps). The EPW is written in local *standard* time, a fixed offset with
43
+ no daylight saving, taken from ``location["utc_offset"]``. Whole-hour
44
+ differences between the two clocks are handled; fractional shifts are not.
45
+ Data finer than an hour (1, 5, 10, 15 or 30 minutes) is aggregated to hours:
46
+ means for intensive quantities, sums for ``precip_mm`` and ``sunshine_min``, a
47
+ speed-weighted vector mean for wind direction. An hour with any sub-interval
48
+ missing is treated as missing.
49
+
50
+ What is derived
51
+ ---------------
52
+ Direct and diffuse irradiance are rarely measured. When ``dhi_Wm2`` and
53
+ ``dni_Wm2`` are absent they are derived from the measured global irradiance:
54
+
55
+ * with ``sunshine_min`` and 10-minute data, from a lookup table of diffuse
56
+ fraction against clearness index, sunshine fraction, sub-hourly clearness
57
+ variability and solar zenith (``"table"``);
58
+ * with hourly ``sunshine_min``, the same without the variability term;
59
+ * otherwise with the DIRINT model from pvlib (``"dirint"``).
60
+
61
+ The tables were fitted on MeteoSwiss Swiss Plateau stations; see
62
+ ``scripts/build_amy_diffuse_table.py``. Outside that climate use ``"dirint"``
63
+ or supply measured irradiance components. Total sky cover is derived from the
64
+ clear-sky index in daylight and from downwelling longwave (calibrated against
65
+ the daylight estimate) at other times. Opaque sky cover equals total sky cover.
66
+ Fields that no station measures are written with the EPW missing codes.
67
+
68
+ Short gaps are interpolated and long gaps raise an error (see
69
+ ``max_gap_hours`` and ``long_gap``). 29 February is dropped, as in every EPW.
70
+ """
71
+
72
+ import datetime as _dt
73
+ import logging
74
+ import unicodedata
75
+ from dataclasses import dataclass, field
76
+ from functools import lru_cache
77
+ from importlib.resources import files
78
+ from typing import Dict, List, Optional, Tuple
79
+
80
+ import numpy as np
81
+ import pandas as pd
82
+ import pvlib
83
+
84
+ from pyepwmorph.tools import psychrometrics
85
+ from pyepwmorph.tools.io import EPW_COLUMN_NAMES
86
+
87
+ logger = logging.getLogger(__name__)
88
+
89
+ #: Column contract. ``agg`` is how sub-hourly values become hourly values.
90
+ COLUMN_SPEC: Dict[str, Dict[str, object]] = {
91
+ "temp_C": dict(unit="degC", agg="mean", required=True, description="Dry-bulb air temperature at 2 m"),
92
+ "dewpoint_C": dict(unit="degC", agg="mean", required="dewpoint_C or rh_pct", description="Dew point"),
93
+ "rh_pct": dict(unit="%", agg="mean", required="dewpoint_C or rh_pct", description="Relative humidity"),
94
+ "pressure_Pa": dict(unit="Pa", agg="mean", required=False, description="Station pressure"),
95
+ "ghi_Wm2": dict(unit="W/m2", agg="mean", required=True, description="Global horizontal irradiance"),
96
+ "dhi_Wm2": dict(unit="W/m2", agg="mean", required=False, description="Measured diffuse horizontal irradiance"),
97
+ "dni_Wm2": dict(unit="W/m2", agg="mean", required=False, description="Measured direct normal irradiance"),
98
+ "lw_down_Wm2": dict(unit="W/m2", agg="mean", required=False, description="Downwelling longwave radiation"),
99
+ "sunshine_min": dict(unit="minutes", agg="sum", required=False, description="Sunshine duration in the interval"),
100
+ "wind_speed_ms": dict(unit="m/s", agg="mean", required=True, description="Wind speed"),
101
+ "wind_dir_deg": dict(unit="degrees", agg="vector", required=True, description="Direction wind blows from"),
102
+ "precip_mm": dict(unit="mm", agg="sum", required=False, description="Precipitation depth in the interval"),
103
+ "snow_depth_cm": dict(unit="cm", agg="last", required=False, description="Snow depth at the end of the interval"),
104
+ "total_sky_cover_tenths": dict(unit="tenths", agg="mean", required=False, description="Observed total sky cover"),
105
+ "opaque_sky_cover_tenths": dict(unit="tenths", agg="mean", required=False, description="Observed opaque sky cover"),
106
+ "visibility_km": dict(unit="km", agg="mean", required=False, description="Visibility"),
107
+ "ceiling_height_m": dict(unit="m", agg="mean", required=False, description="Cloud ceiling height"),
108
+ }
109
+
110
+ #: Columns whose gaps are interpolated linearly in time.
111
+ _LINEAR_COLUMNS = ("temp_C", "dewpoint_C", "rh_pct", "pressure_Pa", "wind_speed_ms", "lw_down_Wm2", "snow_depth_cm")
112
+ #: Columns that must have no long gap.
113
+ _REQUIRED_FOR_GAPS = ("temp_C", "dewpoint_C", "rh_pct", "ghi_Wm2", "wind_speed_ms", "wind_dir_deg")
114
+
115
+ #: Largest zenith angle (degrees) at which irradiance is decomposed.
116
+ _MAX_DECOMPOSE_ZENITH = 85.0
117
+ #: Zenith angle (degrees) beyond which the whole hour is dark and irradiance is set to zero.
118
+ _DARK_ZENITH = 98.0
119
+ #: Fraction of the extraterrestrial normal irradiance DNI may not exceed.
120
+ _MAX_DNI_FRACTION = 0.9
121
+
122
+ #: EnergyPlus missing codes for the fields this module cannot fill.
123
+ _MISSING = dict(
124
+ illuminance=999999, zenith_luminance=9999, visibility=9999.0, ceiling=99999, weather_obs=9,
125
+ weather_codes=999999999, precipitable_water=999, aerosol=0.999, snow_depth=999, days_since_snow=99,
126
+ albedo=999, precip_depth=999, precip_rate=99, horizontal_ir=9999, sky_cover=99,
127
+ )
128
+
129
+ #: Data source and uncertainty flags written to every row.
130
+ _DATA_FLAGS = "?9?9?9?9E0?9?9?9?9?9?9?9?9?9?9?9?9?9?9?9*9*9?9?9?9"
131
+
132
+
133
+ def describe_columns() -> pd.DataFrame:
134
+ """Return the input column contract as a table (name, unit, aggregation, required)."""
135
+ rows = [
136
+ dict(column=name, unit=spec["unit"], aggregation=spec["agg"], required=spec["required"],
137
+ description=spec["description"])
138
+ for name, spec in COLUMN_SPEC.items()
139
+ ]
140
+ return pd.DataFrame(rows).set_index("column")
141
+
142
+
143
+ @dataclass
144
+ class AmyReport:
145
+ """What was done to build an AMY: filled gaps, methods, caveats."""
146
+
147
+ year: int = 0
148
+ source_step_minutes: int = 60
149
+ filled_hours: Dict[str, int] = field(default_factory=dict)
150
+ decomposition: Dict[str, int] = field(default_factory=dict)
151
+ dni_capped_hours: int = 0
152
+ sky_cover_method: str = ""
153
+ sky_cover_calibration: Dict[str, float] = field(default_factory=dict)
154
+ notes: List[str] = field(default_factory=list)
155
+
156
+ def to_text(self) -> str:
157
+ lines = [f"AMY {self.year}: source resolution {self.source_step_minutes} min"]
158
+ if self.filled_hours:
159
+ lines.append("filled hours: " + ", ".join(f"{k}={v}" for k, v in sorted(self.filled_hours.items())))
160
+ if self.decomposition:
161
+ lines.append("diffuse/direct split (hours): " + ", ".join(f"{k}={v}" for k, v in self.decomposition.items()))
162
+ if self.sky_cover_method:
163
+ lines.append(f"sky cover: {self.sky_cover_method}")
164
+ lines.extend(self.notes)
165
+ return "\n".join(lines)
166
+
167
+
168
+ # --------------------------------------------------------------------------- input handling
169
+
170
+
171
+ def _validate_table(table: pd.DataFrame) -> pd.DataFrame:
172
+ if not isinstance(table.index, pd.DatetimeIndex):
173
+ raise ValueError("table needs a DatetimeIndex (see the module docstring for the datetime convention)")
174
+ t = table.copy()
175
+ if t.index.tz is not None:
176
+ t.index = t.index.tz_convert("UTC").tz_localize(None)
177
+ logger.warning("tz-aware index converted to UTC; pass table_utc_offset=0")
178
+ t = t[~t.index.isna()].sort_index()
179
+ if t.index.has_duplicates:
180
+ logger.warning("duplicate timestamps: keeping the first of each")
181
+ t = t[~t.index.duplicated(keep="first")]
182
+ known = [c for c in t.columns if c in COLUMN_SPEC]
183
+ ignored = [c for c in t.columns if c not in COLUMN_SPEC]
184
+ if ignored:
185
+ logger.info("ignoring unknown columns: %s", ", ".join(map(str, ignored)))
186
+ t = t[known].apply(pd.to_numeric, errors="coerce")
187
+ t = t.loc[:, t.notna().any()]
188
+ for name in ("temp_C", "ghi_Wm2", "wind_speed_ms", "wind_dir_deg"):
189
+ if name not in t.columns:
190
+ raise ValueError(f"required column '{name}' is missing or empty")
191
+ if "dewpoint_C" not in t.columns and "rh_pct" not in t.columns:
192
+ raise ValueError("one of 'dewpoint_C' or 'rh_pct' is required")
193
+ return t
194
+
195
+
196
+ def _infer_step_minutes(index: pd.DatetimeIndex) -> int:
197
+ diffs = pd.Series(index[1:] - index[:-1])
198
+ step = int(round(diffs.mode().iloc[0].total_seconds() / 60))
199
+ if step <= 0 or step > 60 or 60 % step != 0:
200
+ raise ValueError(f"unsupported time step of {step} minutes; use 1, 5, 10, 15, 30 or 60")
201
+ return step
202
+
203
+
204
+ def _vector_mean_direction(speed: Optional[pd.Series], direction: pd.Series, k: int) -> pd.Series:
205
+ """Speed-weighted vector mean of a wind direction over *k* sub-intervals."""
206
+ rad = np.radians(direction)
207
+ weight = speed if speed is not None else pd.Series(1.0, index=direction.index)
208
+ calm = weight == 0
209
+ u = (weight * np.sin(rad)).where(~calm, 0.0).where(direction.notna() | calm)
210
+ v = (weight * np.cos(rad)).where(~calm, 0.0).where(direction.notna() | calm)
211
+ mu = u.rolling(k, min_periods=k).mean()
212
+ mv = v.rolling(k, min_periods=k).mean()
213
+ out = np.degrees(np.arctan2(mu, mv)) % 360.0
214
+ out = out.where(np.hypot(mu, mv) > 1e-9, 0.0)
215
+ return out.where(mu.notna() & mv.notna())
216
+
217
+
218
+ def _subhourly_kt_std(ghi: pd.Series, location: dict, step: int) -> pd.Series:
219
+ """Standard deviation of the clearness index across the sub-intervals of each hour.
220
+
221
+ *ghi* is indexed by the interval end in local standard time. The result is
222
+ indexed the same way and is NaN where any sub-interval is missing or the
223
+ sun is low (zenith 85 degrees or more).
224
+ """
225
+ k = 60 // step
226
+ solar = _solar_frame(ghi.index - pd.Timedelta(minutes=step), location, interval_min=step)
227
+ kt = (ghi.clip(lower=0).to_numpy() / (solar["dni_extra"].to_numpy() * solar["cosz"].to_numpy()))
228
+ kt = pd.Series(kt, index=ghi.index).where(solar["zenith"].to_numpy() < _MAX_DECOMPOSE_ZENITH)
229
+ return kt.rolling(k, min_periods=k).std()
230
+
231
+
232
+ def aggregate_to_hourly(
233
+ table: pd.DataFrame,
234
+ location: dict,
235
+ *,
236
+ timestamp_label: str = "end",
237
+ table_utc_offset: float = 0.0,
238
+ ) -> Tuple[pd.DataFrame, int]:
239
+ """Aggregate a station table to hourly rows labelled by the interval *start* in local standard time.
240
+
241
+ Returns the hourly frame and the source time step in minutes. A column
242
+ ``ghi_kt_std`` (sub-hourly clearness variability) is added when the source
243
+ is finer than an hour and has global irradiance.
244
+ """
245
+ if timestamp_label not in ("end", "start"):
246
+ raise ValueError("timestamp_label must be 'end' or 'start'")
247
+ t = _validate_table(table)
248
+ step = _infer_step_minutes(t.index) if len(t) > 1 else 60
249
+ shift = float(location["utc_offset"]) - float(table_utc_offset)
250
+ if abs(shift - round(shift)) > 1e-9:
251
+ raise ValueError(
252
+ "the table's clock and the EPW time zone differ by a fractional number of hours; "
253
+ "convert the table to local standard time first"
254
+ )
255
+ idx = t.index
256
+ if timestamp_label == "start":
257
+ idx = idx + pd.Timedelta(minutes=step)
258
+ t.index = idx + pd.Timedelta(hours=int(round(shift)))
259
+ grid = pd.date_range(t.index.min(), t.index.max(), freq=f"{step}min")
260
+ t = t.reindex(grid)
261
+ if t.index[0].minute != 0 and step == 60:
262
+ raise ValueError("hourly timestamps must fall on the hour")
263
+ k = 60 // step
264
+
265
+ if k == 1:
266
+ hourly = t.copy()
267
+ else:
268
+ parts = {}
269
+ speed = t["wind_speed_ms"] if "wind_speed_ms" in t.columns else None
270
+ for name in t.columns:
271
+ how = COLUMN_SPEC[name]["agg"]
272
+ if how == "mean":
273
+ parts[name] = t[name].rolling(k, min_periods=k).mean()
274
+ elif how == "sum":
275
+ parts[name] = t[name].rolling(k, min_periods=k).sum()
276
+ elif how == "vector":
277
+ parts[name] = _vector_mean_direction(speed, t[name], k)
278
+ else:
279
+ parts[name] = t[name]
280
+ hourly = pd.DataFrame(parts)
281
+ if "ghi_Wm2" in t.columns:
282
+ hourly["ghi_kt_std"] = _subhourly_kt_std(t["ghi_Wm2"], location, step)
283
+ hourly = hourly[hourly.index.minute == 0]
284
+
285
+ hourly.index = hourly.index - pd.Timedelta(hours=1)
286
+ return hourly, step
287
+
288
+
289
+ # --------------------------------------------------------------------------- solar helpers
290
+
291
+
292
+ def _solar_frame(index_start: pd.DatetimeIndex, location: dict, interval_min: int = 60) -> pd.DataFrame:
293
+ """Solar position and extraterrestrial radiation at the middle of each interval.
294
+
295
+ *index_start* holds interval starts as tz-naive local standard time.
296
+ """
297
+ tz = _dt.timezone(_dt.timedelta(hours=float(location["utc_offset"])))
298
+ mid = (index_start + pd.Timedelta(minutes=interval_min / 2.0)).tz_localize(tz)
299
+ pos = pvlib.solarposition.get_solarposition(
300
+ mid, float(location["latitude"]), float(location["longitude"]), altitude=float(location["elevation"])
301
+ )
302
+ dni_extra = np.asarray(pvlib.irradiance.get_extra_radiation(mid), dtype=float)
303
+ zen = pos["zenith"].to_numpy()
304
+ cosz = np.cos(np.radians(np.minimum(zen, 89.0)))
305
+ return pd.DataFrame(
306
+ {
307
+ "zenith": zen,
308
+ "apparent_zenith": pos["apparent_zenith"].to_numpy(),
309
+ "cosz": cosz,
310
+ "dni_extra": dni_extra,
311
+ "ext_hor": dni_extra * np.maximum(np.cos(np.radians(zen)), 0.0),
312
+ },
313
+ index=index_start,
314
+ )
315
+
316
+
317
+ def decomposition_features(
318
+ ghi: pd.Series, sunshine_min: Optional[pd.Series], kt_std: Optional[pd.Series], solar: pd.DataFrame
319
+ ) -> pd.DataFrame:
320
+ """Hourly features of the diffuse-fraction tables (also used to fit them).
321
+
322
+ Columns ``kt`` (clearness index), ``S`` (sunshine fraction of the hour),
323
+ ``sd`` (sub-hourly clearness standard deviation) and ``zen`` (zenith, degrees).
324
+ """
325
+ kt = (ghi.clip(lower=0) / (solar["dni_extra"] * solar["cosz"])).clip(0.0, 1.2)
326
+ feats = pd.DataFrame({"kt": kt, "zen": solar["zenith"]}, index=ghi.index)
327
+ feats["S"] = (sunshine_min / 60.0).clip(0.0, 1.0) if sunshine_min is not None else np.nan
328
+ feats["sd"] = kt_std.clip(0.0, 0.5) if kt_std is not None else np.nan
329
+ return feats[["kt", "S", "sd", "zen"]]
330
+
331
+
332
+ @lru_cache(maxsize=1)
333
+ def _load_diffuse_tables() -> Dict[str, Tuple[List[str], list, np.ndarray]]:
334
+ resource = files("pyepwmorph.data").joinpath("amy_diffuse_table.parquet")
335
+ with resource.open("rb") as handle:
336
+ raw = pd.read_parquet(handle)
337
+ tables = {}
338
+ for name, frame in raw.groupby("table"):
339
+ axes_names = ["kt", "S", "sd", "zen"] if name == "kssd" else ["kt", "S", "zen"]
340
+ axes = [np.sort(frame[a].unique()) for a in axes_names]
341
+ frame = frame.sort_values(axes_names)
342
+ values = frame["kd"].to_numpy().reshape([len(a) for a in axes])
343
+ tables[name] = (axes_names, axes, values)
344
+ return tables
345
+
346
+
347
+ def _table_diffuse_fraction(feats: pd.DataFrame, name: str) -> pd.Series:
348
+ from scipy.interpolate import RegularGridInterpolator
349
+
350
+ axes_names, axes, values = _load_diffuse_tables()[name]
351
+ interp = RegularGridInterpolator(axes, values, bounds_error=False, fill_value=None)
352
+ pts = np.column_stack([
353
+ feats[a].clip(lower=ax[0], upper=ax[-1]).to_numpy() for a, ax in zip(axes_names, axes)
354
+ ])
355
+ return pd.Series(np.clip(interp(pts), 0.0, 1.0), index=feats.index)
356
+
357
+
358
+ def _decompose(
359
+ ghi: pd.Series,
360
+ hourly: pd.DataFrame,
361
+ solar: pd.DataFrame,
362
+ pressure_pa: pd.Series,
363
+ step: int,
364
+ method: str,
365
+ report: AmyReport,
366
+ ) -> Tuple[pd.Series, pd.Series]:
367
+ """Return diffuse horizontal and direct normal irradiance (W/m2)."""
368
+ if method not in ("auto", "table", "dirint", "erbs"):
369
+ raise ValueError("decomposition must be 'auto', 'table', 'dirint' or 'erbs'")
370
+ n = len(ghi)
371
+ dhi = pd.Series(np.nan, index=ghi.index)
372
+ how = pd.Series("", index=ghi.index, dtype=object)
373
+
374
+ daylight = (solar["zenith"] < _MAX_DECOMPOSE_ZENITH) & (ghi > 10.0)
375
+ dark = ~daylight
376
+ dhi[dark] = ghi[dark]
377
+ how[dark] = "night"
378
+
379
+ # measured components first
380
+ cosz = solar["cosz"]
381
+ if "dhi_Wm2" in hourly.columns:
382
+ m = hourly["dhi_Wm2"].notna() & daylight
383
+ dhi[m] = hourly["dhi_Wm2"][m].clip(lower=0).clip(upper=ghi[m])
384
+ how[m] = "measured"
385
+ if "dni_Wm2" in hourly.columns:
386
+ m = hourly["dni_Wm2"].notna() & daylight & dhi.isna()
387
+ dhi[m] = (ghi[m] - hourly["dni_Wm2"][m] * cosz[m]).clip(lower=0).clip(upper=ghi[m])
388
+ how[m] = "measured"
389
+
390
+ todo = dhi.isna() & daylight
391
+ sunshine = hourly["sunshine_min"] if "sunshine_min" in hourly.columns else None
392
+ kt_std = hourly["ghi_kt_std"] if "ghi_kt_std" in hourly.columns else None
393
+ feats = decomposition_features(ghi, sunshine, kt_std, solar)
394
+
395
+ use_tables = method in ("auto", "table") and sunshine is not None
396
+ if method == "table" and sunshine is None:
397
+ raise ValueError("decomposition='table' needs the 'sunshine_min' column")
398
+ if use_tables:
399
+ has_s = feats["S"].notna()
400
+ if step == 10:
401
+ m = todo & has_s & feats["sd"].notna()
402
+ if m.any():
403
+ dhi[m] = _table_diffuse_fraction(feats[m], "kssd") * ghi[m]
404
+ how[m] = "table (sunshine, variability)"
405
+ m = dhi.isna() & todo & has_s
406
+ if m.any():
407
+ dhi[m] = _table_diffuse_fraction(feats[m], "ks") * ghi[m]
408
+ how[m] = "table (sunshine)"
409
+
410
+ todo = dhi.isna() & daylight
411
+ if todo.any():
412
+ times = ghi.index + pd.Timedelta(minutes=30)
413
+ label = "erbs" if method == "erbs" else "dirint"
414
+ if label == "dirint":
415
+ tz = _dt.timezone(_dt.timedelta(hours=0))
416
+ dni = pvlib.irradiance.dirint(
417
+ ghi.to_numpy(), solar["zenith"].to_numpy(), times.tz_localize(tz), pressure=pressure_pa.to_numpy()
418
+ )
419
+ dni = pd.Series(np.asarray(dni, dtype=float), index=ghi.index)
420
+ fallback = (ghi - dni * cosz).clip(lower=0)
421
+ ok = todo & dni.notna()
422
+ dhi[ok] = fallback[ok]
423
+ how[ok] = "dirint"
424
+ todo = dhi.isna() & daylight
425
+ if todo.any():
426
+ er = pvlib.irradiance.erbs(ghi[todo], solar["zenith"][todo], pd.DatetimeIndex(ghi.index[todo]))
427
+ dhi[todo] = er["dhi"].to_numpy()
428
+ how[todo] = "erbs"
429
+
430
+ dhi = dhi.clip(lower=0.0)
431
+ dhi = pd.Series(np.minimum(dhi.to_numpy(), ghi.to_numpy()), index=ghi.index)
432
+ low_sun = solar["zenith"] >= _MAX_DECOMPOSE_ZENITH
433
+ dni = pd.Series(0.0, index=ghi.index)
434
+ ok = ~low_sun & (ghi > 0)
435
+ dni[ok] = (ghi[ok] - dhi[ok]) / cosz[ok]
436
+ cap = _MAX_DNI_FRACTION * solar["dni_extra"]
437
+ capped = dni > cap
438
+ report.dni_capped_hours = int(capped.sum())
439
+ if capped.any():
440
+ dni[capped] = cap[capped]
441
+ dhi[capped] = ghi[capped] - dni[capped] * cosz[capped]
442
+ dni = dni.clip(lower=0.0)
443
+ report.decomposition = {k: int(v) for k, v in how.value_counts().items() if k}
444
+ assert n == len(dni)
445
+ return dhi, dni
446
+
447
+
448
+ # --------------------------------------------------------------------------- sky cover
449
+
450
+
451
+ def estimate_sky_cover(
452
+ ghi: pd.Series,
453
+ solar: pd.DataFrame,
454
+ temp_c: pd.Series,
455
+ dewpoint_c: pd.Series,
456
+ lw_down: Optional[pd.Series],
457
+ location: dict,
458
+ report: AmyReport,
459
+ ) -> pd.Series:
460
+ """Total sky cover as a 0-1 fraction.
461
+
462
+ Daylight (zenith below 75 degrees): Kasten and Czeplak (1980) from the
463
+ ratio of measured to Ineichen clear-sky irradiance. Otherwise, from the
464
+ effective sky emissivity of the downwelling longwave: clear-sky emissivity
465
+ from Martin and Berdahl (1984), cloud fraction as the share of the way to
466
+ an overcast sky, then a straight-line fit to the daylight estimate removes
467
+ the site's bias. Without longwave data, night values are carried across
468
+ from daylight by interpolation.
469
+ """
470
+ mid = (ghi.index + pd.Timedelta(minutes=30)).tz_localize(_dt.timezone(_dt.timedelta(hours=float(location["utc_offset"]))))
471
+ try:
472
+ linke = pvlib.clearsky.lookup_linke_turbidity(mid, float(location["latitude"]), float(location["longitude"]))
473
+ linke = np.asarray(linke, dtype=float)
474
+ except Exception as exc: # the lookup needs pvlib's bundled turbidity file
475
+ logger.warning("Linke turbidity lookup failed (%s); using 3.0", exc)
476
+ linke = np.full(len(mid), 3.0)
477
+ loc = pvlib.location.Location(float(location["latitude"]), float(location["longitude"]),
478
+ altitude=float(location["elevation"]))
479
+ clear = loc.get_clearsky(mid, model="ineichen", linke_turbidity=linke)["ghi"].to_numpy()
480
+ day = (solar["zenith"] < 75.0).to_numpy()
481
+ kc = np.where(day & (clear > 50), ghi.clip(lower=0).to_numpy() / np.where(clear > 50, clear, np.nan), np.nan)
482
+ n_ghi = pd.Series(((1.0 - np.clip(kc, None, 1.0)) / 0.75).clip(0, 1) ** (1 / 3.4), index=ghi.index)
483
+
484
+ n_lw = None
485
+ if lw_down is not None:
486
+ sigma = 5.670374419e-8
487
+ eps = lw_down / (sigma * (temp_c + 273.15) ** 4)
488
+ solar_hour = ((mid.hour + 0.5).to_numpy() + (float(location["longitude"]) - 15.0 * float(location["utc_offset"])) / 15.0)
489
+ eps_clear = 0.711 + 0.0056 * dewpoint_c + 0.000073 * dewpoint_c ** 2 + 0.013 * np.cos(2 * np.pi * solar_hour / 24.0)
490
+ n_lw = ((eps - eps_clear) / (1.0 - eps_clear)).clip(0, 1)
491
+
492
+ cover = n_ghi.copy()
493
+ if n_lw is not None:
494
+ pair = n_ghi.notna() & n_lw.notna() & (solar["zenith"] < 70.0)
495
+ method = "longwave"
496
+ if pair.sum() >= 200 and n_ghi[pair].corr(n_lw[pair]) > 0.5:
497
+ slope, intercept = np.polyfit(n_lw[pair], n_ghi[pair], 1)
498
+ if 0.5 <= slope <= 3.0:
499
+ n_lw = (intercept + slope * n_lw).clip(0, 1)
500
+ method = "longwave calibrated to daylight clear-sky index"
501
+ report.sky_cover_calibration = dict(
502
+ slope=float(slope), intercept=float(intercept),
503
+ correlation=float(n_ghi[pair].corr(n_lw[pair])), hours=int(pair.sum()),
504
+ )
505
+ cover = n_ghi.where(n_ghi.notna(), n_lw)
506
+ report.sky_cover_method = f"daylight: clear-sky index (Kasten & Czeplak 1980); otherwise {method} (Martin & Berdahl 1984)"
507
+ else:
508
+ cover = n_ghi.interpolate(limit=24, limit_direction="both")
509
+ report.sky_cover_method = "daylight: clear-sky index (Kasten & Czeplak 1980); night interpolated, no longwave data"
510
+ return cover
511
+
512
+
513
+ # --------------------------------------------------------------------------- gaps
514
+
515
+
516
+ def _runs(mask: pd.Series) -> List[Tuple[pd.Timestamp, int]]:
517
+ """Start and length of each run of True in *mask*."""
518
+ out = []
519
+ values = mask.to_numpy()
520
+ i = 0
521
+ while i < len(values):
522
+ if values[i]:
523
+ j = i
524
+ while j < len(values) and values[j]:
525
+ j += 1
526
+ out.append((mask.index[i], j - i))
527
+ i = j
528
+ else:
529
+ i += 1
530
+ return out
531
+
532
+
533
+ def _fill_gaps(
534
+ hourly: pd.DataFrame, solar: pd.DataFrame, max_gap_hours: int, long_gap: str, report: AmyReport
535
+ ) -> pd.DataFrame:
536
+ if long_gap not in ("raise", "interpolate"):
537
+ raise ValueError("long_gap must be 'raise' or 'interpolate'")
538
+ h = hourly.copy()
539
+ limit = None if long_gap == "interpolate" else int(max_gap_hours)
540
+
541
+ problems = []
542
+ for name in _REQUIRED_FOR_GAPS:
543
+ if name in h.columns:
544
+ for start, length in _runs(h[name].isna()):
545
+ if length > max_gap_hours:
546
+ problems.append(f"{name}: {length} h from {start:%Y-%m-%d %H:%M}")
547
+ if problems and long_gap == "raise":
548
+ raise ValueError(
549
+ f"gaps longer than {max_gap_hours} h: " + "; ".join(problems[:10])
550
+ + ". Fill them, or pass long_gap='interpolate'."
551
+ )
552
+ if problems:
553
+ report.notes.append("long gaps interpolated: " + "; ".join(problems[:10]))
554
+
555
+ for name in _LINEAR_COLUMNS:
556
+ if name in h.columns:
557
+ before = int(h[name].isna().sum())
558
+ h[name] = h[name].interpolate(method="linear", limit=limit, limit_direction="both")
559
+ after = int(h[name].isna().sum())
560
+ if before - after:
561
+ report.filled_hours[name] = before - after
562
+
563
+ if "wind_dir_deg" in h.columns:
564
+ rad = np.radians(h["wind_dir_deg"])
565
+ u, v = np.sin(rad), np.cos(rad)
566
+ before = int(h["wind_dir_deg"].isna().sum())
567
+ u = u.interpolate(limit=limit, limit_direction="both")
568
+ v = v.interpolate(limit=limit, limit_direction="both")
569
+ h["wind_dir_deg"] = (np.degrees(np.arctan2(u, v)) % 360.0).where(u.notna() & v.notna())
570
+ if before - int(h["wind_dir_deg"].isna().sum()):
571
+ report.filled_hours["wind_dir_deg"] = before - int(h["wind_dir_deg"].isna().sum())
572
+
573
+ # irradiance: interpolate the clearness index so a gap keeps the sun's shape
574
+ ghi = h["ghi_Wm2"].clip(lower=0)
575
+ denom = solar["dni_extra"] * solar["cosz"]
576
+ high_sun = solar["zenith"] < _MAX_DECOMPOSE_ZENITH
577
+ before = int(ghi.isna().sum())
578
+ kt_filled = (ghi / denom).where(high_sun).interpolate(limit=limit, limit_direction="both")
579
+ ghi = (kt_filled * denom).where(high_sun, ghi.interpolate(limit=limit, limit_direction="both"))
580
+ # an hour is dark when the sun stays below the horizon for all of it (the sun moves at most 8 degrees)
581
+ ghi = ghi.where(solar["zenith"] < _DARK_ZENITH, 0.0)
582
+ h["ghi_Wm2"] = ghi
583
+ if before - int(ghi.isna().sum()):
584
+ report.filled_hours["ghi_Wm2"] = before - int(ghi.isna().sum())
585
+ return h
586
+
587
+
588
+ # --------------------------------------------------------------------------- build
589
+
590
+
591
+ def build_amy_dataframe(
592
+ table: pd.DataFrame,
593
+ location: dict,
594
+ year: int,
595
+ *,
596
+ timestamp_label: str = "end",
597
+ table_utc_offset: float = 0.0,
598
+ max_gap_hours: int = 6,
599
+ long_gap: str = "raise",
600
+ decomposition: str = "auto",
601
+ ) -> Tuple[pd.DataFrame, AmyReport]:
602
+ """Build the 8760 EPW data rows for *year* from a station table.
603
+
604
+ Parameters
605
+ ----------
606
+ table : pd.DataFrame
607
+ Station measurements; see the module docstring for columns and units.
608
+ location : dict
609
+ ``latitude``, ``longitude``, ``elevation`` (m) and ``utc_offset``
610
+ (hours east of UTC, standard time), plus optional ``site``, ``province``,
611
+ ``country_code``, ``type`` and ``usaf`` for the EPW header.
612
+ year : int
613
+ The calendar year, in local standard time.
614
+ timestamp_label : {"end", "start"}
615
+ Which end of each averaging interval the index marks.
616
+ table_utc_offset : float
617
+ Hours east of UTC of the table's clock (0 for UTC).
618
+ max_gap_hours : int
619
+ Longest gap that is interpolated.
620
+ long_gap : {"raise", "interpolate"}
621
+ What to do about longer gaps in the required columns.
622
+ decomposition : {"auto", "table", "dirint", "erbs"}
623
+ How to split global irradiance when no components are measured.
624
+
625
+ Returns
626
+ -------
627
+ tuple
628
+ A data frame with the 35 EPW columns (``hour`` runs 1-24) and an :class:`AmyReport`.
629
+ """
630
+ for key in ("latitude", "longitude", "elevation", "utc_offset"):
631
+ if key not in location:
632
+ raise ValueError(f"location is missing '{key}'")
633
+ report = AmyReport(year=int(year))
634
+ hourly, step = aggregate_to_hourly(table, location, timestamp_label=timestamp_label, table_utc_offset=table_utc_offset)
635
+ report.source_step_minutes = step
636
+
637
+ full = pd.date_range(f"{year}-01-01 00:00", f"{year}-12-31 23:00", freq="h")
638
+ present = hourly.index.intersection(full)
639
+ if len(present) < 0.5 * len(full):
640
+ raise ValueError(f"the table covers only {len(present)} of the {len(full)} hours of {year}")
641
+ hourly = hourly.reindex(full)
642
+ leap_day = (full.month == 2) & (full.day == 29)
643
+ if leap_day.any():
644
+ hourly = hourly[~leap_day]
645
+ report.notes.append("29 February dropped")
646
+
647
+ solar = _solar_frame(hourly.index, location)
648
+ hourly = _fill_gaps(hourly, solar, max_gap_hours, long_gap, report)
649
+
650
+ for name in ("temp_C", "ghi_Wm2", "wind_speed_ms", "wind_dir_deg"):
651
+ left = int(hourly[name].isna().sum())
652
+ if left:
653
+ raise ValueError(f"{left} hours of '{name}' are still missing after filling")
654
+
655
+ temp = hourly["temp_C"]
656
+ if "dewpoint_C" in hourly.columns and "rh_pct" in hourly.columns:
657
+ dew, rh = hourly["dewpoint_C"], hourly["rh_pct"]
658
+ elif "dewpoint_C" in hourly.columns:
659
+ dew = hourly["dewpoint_C"]
660
+ rh = 100.0 * pd.Series(
661
+ psychrometrics.saturated_vapor_pressure(dew + 273.15) / psychrometrics.saturated_vapor_pressure(temp + 273.15),
662
+ index=hourly.index,
663
+ )
664
+ else:
665
+ rh = hourly["rh_pct"].clip(0, 100)
666
+ dew = pd.Series(psychrometrics.dew_point_from_db_rh(temp.to_numpy(), rh.to_numpy()), index=hourly.index)
667
+ if dew.isna().any() or rh.isna().any():
668
+ raise ValueError("humidity is still missing after filling")
669
+ rh = rh.clip(0, 100)
670
+ dew = np.minimum(dew, temp)
671
+
672
+ if "pressure_Pa" in hourly.columns and hourly["pressure_Pa"].notna().all():
673
+ pressure = hourly["pressure_Pa"]
674
+ else:
675
+ std = float(pvlib.atmosphere.alt2pres(float(location["elevation"])))
676
+ pressure = pd.Series(std, index=hourly.index)
677
+ report.notes.append("pressure from the standard atmosphere at the station elevation")
678
+
679
+ ghi = hourly["ghi_Wm2"].clip(lower=0)
680
+ dhi, dni = _decompose(ghi, hourly, solar, pressure, step, decomposition, report)
681
+
682
+ lw = hourly["lw_down_Wm2"] if "lw_down_Wm2" in hourly.columns else None
683
+ if "total_sky_cover_tenths" in hourly.columns and hourly["total_sky_cover_tenths"].notna().any():
684
+ cover = (hourly["total_sky_cover_tenths"] / 10.0).clip(0, 1)
685
+ report.sky_cover_method = "observed"
686
+ else:
687
+ cover = estimate_sky_cover(ghi, solar, temp, dew, lw, location, report)
688
+ tenths = np.rint(cover * 10.0)
689
+ total = tenths.where(tenths.notna(), _MISSING["sky_cover"]).astype(int)
690
+ if "opaque_sky_cover_tenths" in hourly.columns and hourly["opaque_sky_cover_tenths"].notna().all():
691
+ opaque = hourly["opaque_sky_cover_tenths"].round().astype(int)
692
+ else:
693
+ opaque = total
694
+
695
+ start = hourly.index
696
+ n = len(hourly)
697
+
698
+ def col(name, default):
699
+ if name in hourly.columns:
700
+ return hourly[name].fillna(default)
701
+ return pd.Series(default, index=start)
702
+
703
+ wind_speed = hourly["wind_speed_ms"].clip(lower=0)
704
+ wind_dir = hourly["wind_dir_deg"].where(wind_speed > 0, 0.0).round()
705
+
706
+ epw = pd.DataFrame(index=start, columns=EPW_COLUMN_NAMES, dtype=object)
707
+ epw["year"] = int(year)
708
+ epw["month"] = start.month
709
+ epw["day"] = start.day
710
+ epw["hour"] = start.hour + 1
711
+ epw["minute"] = 0
712
+ epw["datasource"] = _DATA_FLAGS
713
+ epw["drybulb_C"] = temp.round(1)
714
+ epw["dewpoint_C"] = dew.round(1)
715
+ epw["relhum_percent"] = rh.round(1)
716
+ epw["atmos_Pa"] = pressure.round(0).astype(int)
717
+ epw["exthorrad_Whm2"] = solar["ext_hor"].round(0).astype(int)
718
+ epw["extdirrad_Whm2"] = solar["dni_extra"].round(0).astype(int)
719
+ epw["horirsky_Whm2"] = (
720
+ lw.round(0).astype(int) if lw is not None and lw.notna().all() else _MISSING["horizontal_ir"]
721
+ )
722
+ epw["glohorrad_Whm2"] = ghi.round(0).astype(int)
723
+ epw["dirnorrad_Whm2"] = dni.round(0).astype(int)
724
+ epw["difhorrad_Whm2"] = dhi.round(0).astype(int)
725
+ epw["glohorillum_lux"] = _MISSING["illuminance"]
726
+ epw["dirnorillum_lux"] = _MISSING["illuminance"]
727
+ epw["difhorillum_lux"] = _MISSING["illuminance"]
728
+ epw["zenlum_lux"] = _MISSING["zenith_luminance"]
729
+ epw["winddir_deg"] = wind_dir.astype(int)
730
+ epw["windspd_ms"] = wind_speed.round(1)
731
+ epw["totskycvr_tenths"] = total
732
+ epw["opaqskycvr_tenths"] = opaque
733
+ epw["visibility_km"] = (
734
+ col("visibility_km", _MISSING["visibility"]).round(1) if "visibility_km" in hourly.columns else _MISSING["visibility"]
735
+ )
736
+ epw["ceiling_hgt_m"] = (
737
+ col("ceiling_height_m", _MISSING["ceiling"]).round(0).astype(int)
738
+ if "ceiling_height_m" in hourly.columns else _MISSING["ceiling"]
739
+ )
740
+ epw["presweathobs"] = _MISSING["weather_obs"]
741
+ epw["presweathcodes"] = _MISSING["weather_codes"]
742
+ epw["precip_wtr_mm"] = _MISSING["precipitable_water"]
743
+ epw["aerosol_opt_thousandths"] = _MISSING["aerosol"]
744
+ epw["snowdepth_cm"] = (
745
+ col("snow_depth_cm", _MISSING["snow_depth"]).clip(lower=0).round(0).astype(int)
746
+ if "snow_depth_cm" in hourly.columns else _MISSING["snow_depth"]
747
+ )
748
+ epw["days_last_snow"] = _MISSING["days_since_snow"]
749
+ epw["Albedo"] = _MISSING["albedo"]
750
+ if "precip_mm" in hourly.columns:
751
+ precip = hourly["precip_mm"]
752
+ epw["liq_precip_depth_mm"] = precip.clip(lower=0).round(1).where(precip.notna(), _MISSING["precip_depth"])
753
+ epw["liq_precip_rate_Hour"] = np.where(precip.notna(), 1.0, _MISSING["precip_rate"])
754
+ else:
755
+ epw["liq_precip_depth_mm"] = _MISSING["precip_depth"]
756
+ epw["liq_precip_rate_Hour"] = _MISSING["precip_rate"]
757
+ assert len(epw) == n == 8760
758
+ return epw, report
759
+
760
+
761
+ # --------------------------------------------------------------------------- output
762
+
763
+ _FORMATS = {
764
+ "year": "{:d}", "month": "{:d}", "day": "{:d}", "hour": "{:d}", "minute": "{:d}",
765
+ "drybulb_C": "{:.1f}", "dewpoint_C": "{:.1f}", "relhum_percent": "{:.1f}", "atmos_Pa": "{:d}",
766
+ "windspd_ms": "{:.1f}", "visibility_km": "{:.1f}", "liq_precip_depth_mm": "{:.1f}",
767
+ "liq_precip_rate_Hour": "{:.1f}", "aerosol_opt_thousandths": "{:.3f}",
768
+ }
769
+
770
+
771
+ def _ascii(value) -> str:
772
+ """Plain-ASCII text for header fields (EnergyPlus is happiest without accents)."""
773
+ return unicodedata.normalize("NFKD", str(value)).encode("ascii", "ignore").decode("ascii")
774
+
775
+
776
+ def _format_rows(epw: pd.DataFrame) -> List[str]:
777
+ cols = []
778
+ for name in EPW_COLUMN_NAMES:
779
+ values = epw[name].tolist()
780
+ if name == "datasource":
781
+ cols.append([str(v) for v in values])
782
+ continue
783
+ fmt = _FORMATS.get(name, "{:d}")
784
+ cols.append([fmt.format(float(v) if "f}" in fmt else int(v)) for v in values])
785
+ return [",".join(row) + "\n" for row in zip(*cols)]
786
+
787
+
788
+ def write_amy_epw(
789
+ path: str,
790
+ epw: pd.DataFrame,
791
+ location: dict,
792
+ report: AmyReport,
793
+ *,
794
+ source_name: str = "weather station",
795
+ attribution: Optional[str] = None,
796
+ ) -> None:
797
+ """Write the data rows from :func:`build_amy_dataframe` to an EPW file.
798
+
799
+ The header states the period of record (``Period of Record=<year>-<year>``)
800
+ and the methods used for derived fields, so downstream tools that read the
801
+ baseline period see a single year.
802
+ """
803
+ year = int(epw["year"].iloc[0])
804
+ first = _dt.date(year, 1, 1)
805
+ day_of_week = first.strftime("%A")
806
+
807
+ def text(key, default):
808
+ return _ascii(location.get(key, default)).replace(",", " ")
809
+
810
+ loc_line = ",".join([
811
+ "LOCATION", text("site", "Unknown"), text("province", "-"), text("country_code", "-"),
812
+ text("type", "Measured"), text("usaf", "999999"),
813
+ f"{float(location['latitude']):.5f}", f"{float(location['longitude']):.5f}",
814
+ f"{float(location['utc_offset']):.1f}", f"{float(location['elevation']):.1f}",
815
+ ])
816
+
817
+ methods = []
818
+ if report.decomposition:
819
+ methods.append("direct/diffuse split: " + ", ".join(f"{k} {v} h" for k, v in report.decomposition.items()))
820
+ if report.sky_cover_method:
821
+ methods.append("sky cover: " + report.sky_cover_method)
822
+ if report.filled_hours:
823
+ methods.append("interpolated hours: " + ", ".join(f"{k} {v}" for k, v in sorted(report.filled_hours.items())))
824
+ comments_2 = (
825
+ f"Actual meteorological year {year}; hourly means in local standard time, hour ending; "
826
+ + "; ".join(methods)
827
+ + ("; " + attribution if attribution else "")
828
+ ).replace('"', "'")
829
+ comments_2 = _ascii(comments_2)
830
+ comments_1 = (
831
+ f"Measured data from {source_name}, built with pyepwmorph; Period of Record={year}-{year}"
832
+ ).replace('"', "'")
833
+ comments_1 = _ascii(comments_1)
834
+
835
+ header = [
836
+ loc_line,
837
+ "DESIGN CONDITIONS,0",
838
+ "TYPICAL/EXTREME PERIODS,0",
839
+ "GROUND TEMPERATURES,0",
840
+ "HOLIDAYS/DAYLIGHT SAVINGS,No,0,0,0",
841
+ f'COMMENTS 1,"{comments_1}"',
842
+ f'COMMENTS 2,"{comments_2}"',
843
+ f"DATA PERIODS,1,1,Data,{day_of_week},1/1,12/31",
844
+ ]
845
+ with open(path, "w", encoding="utf-8") as fh:
846
+ fh.write("\n".join(header) + "\n")
847
+ fh.writelines(_format_rows(epw))
848
+
849
+
850
+ def station_table_to_epw(
851
+ table: pd.DataFrame,
852
+ location: dict,
853
+ year: int,
854
+ output_path: str,
855
+ *,
856
+ source_name: str = "weather station",
857
+ attribution: Optional[str] = None,
858
+ **build_kwargs,
859
+ ) -> AmyReport:
860
+ """Build an AMY from a station table and write it to *output_path*.
861
+
862
+ Keyword arguments other than ``source_name`` and ``attribution`` go to
863
+ :func:`build_amy_dataframe`. Returns the :class:`AmyReport`.
864
+ """
865
+ epw, report = build_amy_dataframe(table, location, year, **build_kwargs)
866
+ write_amy_epw(output_path, epw, location, report, source_name=source_name, attribution=attribution)
867
+ return report
@@ -0,0 +1,123 @@
1
+ """MeteoSwiss open data (ogd-smn) adapter for :mod:`pyepwmorph.tools.amy`.
2
+
3
+ Reads the automatic station files that MeteoSwiss publishes at
4
+ https://opendatadocs.meteoswiss.ch/ (10-minute ``..._t_...`` and hourly
5
+ ``..._h_...`` CSVs, semicolon separated) and renames their parameter codes to
6
+ the columns :func:`pyepwmorph.tools.amy.build_amy_dataframe` expects. The
7
+ station table is the ``ogd-smn_meta_stations.csv`` file.
8
+
9
+ MeteoSwiss timestamps (``reference_timestamp``, ``DD.MM.YYYY HH:MM``) are UTC
10
+ and mark the *end* of the averaging interval, which is what the defaults of
11
+ ``build_amy_dataframe`` assume.
12
+
13
+ Data: Source MeteoSwiss, open government data (check the current terms for
14
+ attribution wording).
15
+ """
16
+
17
+ import logging
18
+ from typing import Optional
19
+
20
+ import pandas as pd
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+ #: Attribution line for the EPW header.
25
+ ATTRIBUTION = "Source: MeteoSwiss (Federal Office of Meteorology and Climatology), open government data"
26
+
27
+ #: Parameter code -> (amy column, multiplier). 10-minute files (``z``/``s`` suffixes).
28
+ MAPPING_10MIN = {
29
+ "tre200s0": ("temp_C", 1.0),
30
+ "tde200s0": ("dewpoint_C", 1.0),
31
+ "ure200s0": ("rh_pct", 1.0),
32
+ "prestas0": ("pressure_Pa", 100.0),
33
+ "fkl010z0": ("wind_speed_ms", 1.0),
34
+ "dkl010z0": ("wind_dir_deg", 1.0),
35
+ "gre000z0": ("ghi_Wm2", 1.0),
36
+ "ods000z0": ("dhi_Wm2", 1.0),
37
+ "oli000z0": ("lw_down_Wm2", 1.0),
38
+ "sre000z0": ("sunshine_min", 1.0),
39
+ "rre150z0": ("precip_mm", 1.0),
40
+ "htoauts0": ("snow_depth_cm", 1.0),
41
+ }
42
+
43
+ #: Hourly files (``h`` suffix).
44
+ MAPPING_HOURLY = {
45
+ "tre200h0": ("temp_C", 1.0),
46
+ "tde200h0": ("dewpoint_C", 1.0),
47
+ "ure200h0": ("rh_pct", 1.0),
48
+ "prestah0": ("pressure_Pa", 100.0),
49
+ "fkl010h0": ("wind_speed_ms", 1.0),
50
+ "dkl010h0": ("wind_dir_deg", 1.0),
51
+ "gre000h0": ("ghi_Wm2", 1.0),
52
+ "ods000h0": ("dhi_Wm2", 1.0),
53
+ "oli000h0": ("lw_down_Wm2", 1.0),
54
+ "sre000h0": ("sunshine_min", 1.0),
55
+ "rre150h0": ("precip_mm", 1.0),
56
+ "htoauths": ("snow_depth_cm", 1.0),
57
+ }
58
+
59
+ _TIMESTAMP_FORMAT = "%d.%m.%Y %H:%M"
60
+
61
+
62
+ def read_meteoswiss_ogd(path: str, station: Optional[str] = None) -> pd.DataFrame:
63
+ """Read a MeteoSwiss ogd-smn station CSV into the AMY input table.
64
+
65
+ Parameters
66
+ ----------
67
+ path : str
68
+ A 10-minute or hourly station file (decade files such as
69
+ ``ogd-smn_sma_t_historical_2020-2029.csv`` work; so do the ``recent`` files).
70
+ station : str, optional
71
+ Station abbreviation to keep if the file holds several.
72
+
73
+ Returns
74
+ -------
75
+ pd.DataFrame
76
+ Tz-naive UTC timestamps marking the end of each interval, columns named
77
+ as in :data:`pyepwmorph.tools.amy.COLUMN_SPEC` with units converted
78
+ (pressure in Pa).
79
+ """
80
+ raw = pd.read_csv(path, sep=";", encoding="latin-1")
81
+ if "reference_timestamp" not in raw.columns:
82
+ raise ValueError(f"{path} has no 'reference_timestamp' column; is it a MeteoSwiss ogd-smn file?")
83
+ if station is not None and "station_abbr" in raw.columns:
84
+ raw = raw[raw["station_abbr"] == station]
85
+ if any(c in raw.columns for c in MAPPING_10MIN):
86
+ mapping = MAPPING_10MIN
87
+ elif any(c in raw.columns for c in MAPPING_HOURLY):
88
+ mapping = MAPPING_HOURLY
89
+ else:
90
+ raise ValueError(f"{path} has none of the expected MeteoSwiss parameter codes")
91
+ index = pd.to_datetime(raw["reference_timestamp"], format=_TIMESTAMP_FORMAT)
92
+ out = pd.DataFrame(index=index)
93
+ for code, (name, factor) in mapping.items():
94
+ if code in raw.columns:
95
+ out[name] = pd.to_numeric(raw[code], errors="coerce").to_numpy() * factor
96
+ out.index.name = "timestamp_utc_end"
97
+ return out
98
+
99
+
100
+ def meteoswiss_location(stations_csv: str, station: str, utc_offset: float = 1.0) -> dict:
101
+ """EPW location for a station from ``ogd-smn_meta_stations.csv``.
102
+
103
+ ``utc_offset`` is the standard-time offset (1.0 for Switzerland; the data
104
+ stay in fixed standard time, without daylight saving).
105
+ """
106
+ meta = pd.read_csv(stations_csv, sep=";", encoding="latin-1")
107
+ row = meta[meta["station_abbr"] == station]
108
+ if row.empty:
109
+ raise ValueError(f"station '{station}' not found in {stations_csv}")
110
+ row = row.iloc[0]
111
+ wigos = str(row.get("station_wigos_id", ""))
112
+ wmo = wigos.split("-")[-1] if wigos else "999999"
113
+ return dict(
114
+ site=str(row["station_name"]).replace(" / ", "-"),
115
+ province=str(row["station_canton"]),
116
+ country_code="CHE",
117
+ type=f"MeteoSwiss-{station}",
118
+ usaf=wmo,
119
+ latitude=float(row["station_coordinates_wgs84_lat"]),
120
+ longitude=float(row["station_coordinates_wgs84_lon"]),
121
+ utc_offset=float(utc_offset),
122
+ elevation=float(row["station_height_masl"]),
123
+ )
@@ -17,7 +17,7 @@ core-metadata-version = "2.4"
17
17
 
18
18
  [project]
19
19
  name = "pyepwmorph"
20
- version = "3.2.0"
20
+ version = "3.3.0"
21
21
  authors = [
22
22
  { name="Justin McCarty", email="mccarty.justin.f@gmail.com" },
23
23
  ]
File without changes