pyepwmorph 3.2.0__tar.gz → 3.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/.gitignore +4 -0
  2. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/PKG-INFO +10 -1
  3. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/README.md +9 -0
  4. pyepwmorph-3.4.0/pyepwmorph/data/amy_diffuse_table.parquet +0 -0
  5. pyepwmorph-3.4.0/pyepwmorph/tools/amy.py +880 -0
  6. pyepwmorph-3.4.0/pyepwmorph/tools/amy_meteoswiss.py +135 -0
  7. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyproject.toml +1 -1
  8. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/LICENSE +0 -0
  9. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/__init__.py +0 -0
  10. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/data/__init__.py +0 -0
  11. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/data/ch2025_monthly.parquet +0 -0
  12. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/data/ch2025_stations.parquet +0 -0
  13. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/models/__init__.py +0 -0
  14. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/models/access.py +0 -0
  15. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/models/assemble.py +0 -0
  16. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/models/ch2025.py +0 -0
  17. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/models/coordinate.py +0 -0
  18. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/models/custom.py +0 -0
  19. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/morph/__init__.py +0 -0
  20. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/morph/procedures.py +0 -0
  21. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/__init__.py +0 -0
  22. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/cache.py +0 -0
  23. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/configuration.py +0 -0
  24. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/io.py +0 -0
  25. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/psychrometrics.py +0 -0
  26. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/solar.py +0 -0
  27. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/utilities.py +0 -0
  28. {pyepwmorph-3.2.0 → pyepwmorph-3.4.0}/pyepwmorph/tools/workflow.py +0 -0
@@ -159,3 +159,7 @@ cython_debug/
159
159
  # and can be added to the global gitignore or merged into this file. For a more nuclear
160
160
  # option (not recommended) you can uncomment the following to ignore the entire idea folder.
161
161
  .idea/
162
+
163
+ # Example inputs and outputs of examples/run_amy_meteoswiss.py
164
+ examples/amy_data/
165
+ examples/amy_result/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyepwmorph
3
- Version: 3.2.0
3
+ Version: 3.4.0
4
4
  Summary: A python package to enable simple and easy gathering of climate model data and morphing of EPW files
5
5
  Project-URL: Homepage, https://github.com/justinfmccarty/pyepwmorph
6
6
  Project-URL: Issues, https://github.com/justinfmccarty/pyepwmorph/issues
@@ -152,6 +152,15 @@ The signal is the change from the 1991-2020 reference climate to the chosen warm
152
152
 
153
153
  CH2025 data: MeteoSwiss & ETH Zurich (2025), Climate CH2025 - Daily Datasets, CC-BY 4.0, https://doi.org/10.18751/climate/scenarios/ch2025/data/1.0/
154
154
 
155
+ ### Measured data (actual meteorological year)
156
+
157
+ Moved to [weather-file-builder](https://github.com/justinfmccarty/weather_file_builder)
158
+ (2.1 and later), together with ISO 15927-4 typical years from station data:
159
+ `weather_file_builder.amy.station_table_to_epw` and
160
+ `weather_file_builder.station_tmy.station_table_to_tmy_epw`.
161
+ `pyepwmorph.tools.amy` still works in 3.4 but warns on import and will be
162
+ removed in 4.0.
163
+
155
164
  ## Climate scenarios
156
165
 
157
166
  | Scenario | SSP | Description | Expected warming |
@@ -113,6 +113,15 @@ The signal is the change from the 1991-2020 reference climate to the chosen warm
113
113
 
114
114
  CH2025 data: MeteoSwiss & ETH Zurich (2025), Climate CH2025 - Daily Datasets, CC-BY 4.0, https://doi.org/10.18751/climate/scenarios/ch2025/data/1.0/
115
115
 
116
+ ### Measured data (actual meteorological year)
117
+
118
+ Moved to [weather-file-builder](https://github.com/justinfmccarty/weather_file_builder)
119
+ (2.1 and later), together with ISO 15927-4 typical years from station data:
120
+ `weather_file_builder.amy.station_table_to_epw` and
121
+ `weather_file_builder.station_tmy.station_table_to_tmy_epw`.
122
+ `pyepwmorph.tools.amy` still works in 3.4 but warns on import and will be
123
+ removed in 4.0.
124
+
116
125
  ## Climate scenarios
117
126
 
118
127
  | Scenario | SSP | Description | Expected warming |
@@ -0,0 +1,880 @@
1
+ """Build an EPW from measured weather station data (an actual meteorological year).
2
+
3
+ .. deprecated:: 3.4.0
4
+ Moved to ``weather_file_builder.amy`` (weather-file-builder 2.1). This
5
+ copy is frozen and will be removed in pyepwmorph 4.0.
6
+
7
+ A typical meteorological year stitches months from many years. An *actual*
8
+ meteorological year (AMY) is one real calendar year of measurements, which is
9
+ what building energy model calibration needs. This module turns a table of
10
+ station measurements, hourly or finer, into an EPW file.
11
+
12
+ Input table
13
+ -----------
14
+ ``table`` is a ``pandas.DataFrame`` with a tz-naive ``DatetimeIndex`` and the
15
+ columns below. Column names are fixed (see :data:`COLUMN_SPEC` and
16
+ :func:`describe_columns`); an adapter such as
17
+ :mod:`pyepwmorph.tools.amy_meteoswiss` renames a network's own columns.
18
+ Unknown columns are ignored and columns that are entirely empty are dropped.
19
+
20
+ ========================= =========== =====================================================
21
+ column unit meaning
22
+ ========================= =========== =====================================================
23
+ ``temp_C`` degC dry-bulb air temperature (required)
24
+ ``dewpoint_C`` degC dew point (this or ``rh_pct`` required)
25
+ ``rh_pct`` % relative humidity (this or ``dewpoint_C`` required)
26
+ ``pressure_Pa`` Pa station pressure (default: standard atmosphere)
27
+ ``ghi_Wm2`` W/m2 global horizontal irradiance, interval mean (required)
28
+ ``dhi_Wm2``, ``dni_Wm2`` W/m2 measured diffuse / direct normal, if the site has them
29
+ ``lw_down_Wm2`` W/m2 downwelling longwave at the surface, interval mean
30
+ ``sunshine_min`` minutes sunshine duration within the interval (WMO, >=120 W/m2)
31
+ ``wind_speed_ms`` m/s wind speed, interval mean (required)
32
+ ``wind_dir_deg`` degrees direction the wind blows from, 0-360, 0 or 360 = north
33
+ ``precip_mm`` mm precipitation depth within the interval
34
+ ``snow_depth_cm`` cm snow depth at the end of the interval
35
+ ``total_sky_cover_tenths`` tenths 0-10 observed sky cover (derived when absent)
36
+ ``opaque_sky_cover_tenths`` tenths 0-10 observed opaque sky cover (set equal to total when absent)
37
+ ``visibility_km`` km visibility
38
+ ``ceiling_height_m`` m cloud ceiling height
39
+ ========================= =========== =====================================================
40
+
41
+ Datetime convention
42
+ -------------------
43
+ Every value describes an *interval*, not an instant. Say which end of the
44
+ interval the index marks with ``timestamp_label`` (``"end"`` or ``"start"``)
45
+ and which clock it uses with ``table_utc_offset`` (hours east of UTC, 0 for UTC
46
+ timestamps). The EPW is written in local *standard* time, a fixed offset with
47
+ no daylight saving, taken from ``location["utc_offset"]``. Whole-hour
48
+ differences between the two clocks are handled; fractional shifts are not.
49
+ Data finer than an hour (1, 5, 10, 15 or 30 minutes) is aggregated to hours:
50
+ means for intensive quantities, sums for ``precip_mm`` and ``sunshine_min``, a
51
+ speed-weighted vector mean for wind direction. An hour with any sub-interval
52
+ missing is treated as missing.
53
+
54
+ What is derived
55
+ ---------------
56
+ Direct and diffuse irradiance are rarely measured. When ``dhi_Wm2`` and
57
+ ``dni_Wm2`` are absent they are derived from the measured global irradiance:
58
+
59
+ * with ``sunshine_min`` and 10-minute data, from a lookup table of diffuse
60
+ fraction against clearness index, sunshine fraction, sub-hourly clearness
61
+ variability and solar zenith (``"table"``);
62
+ * with hourly ``sunshine_min``, the same without the variability term;
63
+ * otherwise with the DIRINT model from pvlib (``"dirint"``).
64
+
65
+ The tables were fitted on MeteoSwiss Swiss Plateau stations; see
66
+ ``scripts/build_amy_diffuse_table.py``. Outside that climate use ``"dirint"``
67
+ or supply measured irradiance components. Total sky cover is derived from the
68
+ clear-sky index in daylight and from downwelling longwave (calibrated against
69
+ the daylight estimate) at other times. Opaque sky cover equals total sky cover.
70
+ Fields that no station measures are written with the EPW missing codes.
71
+
72
+ Short gaps are interpolated and long gaps raise an error (see
73
+ ``max_gap_hours`` and ``long_gap``). 29 February is dropped, as in every EPW.
74
+ """
75
+
76
+ import datetime as _dt
77
+ import logging
78
+ import unicodedata
79
+ import warnings as _warnings
80
+ from dataclasses import dataclass, field
81
+ from functools import lru_cache
82
+ from importlib.resources import files
83
+ from typing import Dict, List, Optional, Tuple
84
+
85
+ import numpy as np
86
+ import pandas as pd
87
+ import pvlib
88
+
89
+ from pyepwmorph.tools import psychrometrics
90
+ from pyepwmorph.tools.io import EPW_COLUMN_NAMES
91
+
92
+ _warnings.warn(
93
+ "pyepwmorph.tools.amy is deprecated and will be removed in pyepwmorph 4.0. "
94
+ "It has moved to weather-file-builder (pip install 'weather-file-builder>=2.1'): "
95
+ "use weather_file_builder.amy.",
96
+ DeprecationWarning,
97
+ stacklevel=2,
98
+ )
99
+
100
+ logger = logging.getLogger(__name__)
101
+
102
+ #: Column contract. ``agg`` is how sub-hourly values become hourly values.
103
+ COLUMN_SPEC: Dict[str, Dict[str, object]] = {
104
+ "temp_C": dict(unit="degC", agg="mean", required=True, description="Dry-bulb air temperature at 2 m"),
105
+ "dewpoint_C": dict(unit="degC", agg="mean", required="dewpoint_C or rh_pct", description="Dew point"),
106
+ "rh_pct": dict(unit="%", agg="mean", required="dewpoint_C or rh_pct", description="Relative humidity"),
107
+ "pressure_Pa": dict(unit="Pa", agg="mean", required=False, description="Station pressure"),
108
+ "ghi_Wm2": dict(unit="W/m2", agg="mean", required=True, description="Global horizontal irradiance"),
109
+ "dhi_Wm2": dict(unit="W/m2", agg="mean", required=False, description="Measured diffuse horizontal irradiance"),
110
+ "dni_Wm2": dict(unit="W/m2", agg="mean", required=False, description="Measured direct normal irradiance"),
111
+ "lw_down_Wm2": dict(unit="W/m2", agg="mean", required=False, description="Downwelling longwave radiation"),
112
+ "sunshine_min": dict(unit="minutes", agg="sum", required=False, description="Sunshine duration in the interval"),
113
+ "wind_speed_ms": dict(unit="m/s", agg="mean", required=True, description="Wind speed"),
114
+ "wind_dir_deg": dict(unit="degrees", agg="vector", required=True, description="Direction wind blows from"),
115
+ "precip_mm": dict(unit="mm", agg="sum", required=False, description="Precipitation depth in the interval"),
116
+ "snow_depth_cm": dict(unit="cm", agg="last", required=False, description="Snow depth at the end of the interval"),
117
+ "total_sky_cover_tenths": dict(unit="tenths", agg="mean", required=False, description="Observed total sky cover"),
118
+ "opaque_sky_cover_tenths": dict(unit="tenths", agg="mean", required=False, description="Observed opaque sky cover"),
119
+ "visibility_km": dict(unit="km", agg="mean", required=False, description="Visibility"),
120
+ "ceiling_height_m": dict(unit="m", agg="mean", required=False, description="Cloud ceiling height"),
121
+ }
122
+
123
+ #: Columns whose gaps are interpolated linearly in time.
124
+ _LINEAR_COLUMNS = ("temp_C", "dewpoint_C", "rh_pct", "pressure_Pa", "wind_speed_ms", "lw_down_Wm2", "snow_depth_cm")
125
+ #: Columns that must have no long gap.
126
+ _REQUIRED_FOR_GAPS = ("temp_C", "dewpoint_C", "rh_pct", "ghi_Wm2", "wind_speed_ms", "wind_dir_deg")
127
+
128
+ #: Largest zenith angle (degrees) at which irradiance is decomposed.
129
+ _MAX_DECOMPOSE_ZENITH = 85.0
130
+ #: Zenith angle (degrees) beyond which the whole hour is dark and irradiance is set to zero.
131
+ _DARK_ZENITH = 98.0
132
+ #: Fraction of the extraterrestrial normal irradiance DNI may not exceed.
133
+ _MAX_DNI_FRACTION = 0.9
134
+
135
+ #: EnergyPlus missing codes for the fields this module cannot fill.
136
+ _MISSING = dict(
137
+ illuminance=999999, zenith_luminance=9999, visibility=9999.0, ceiling=99999, weather_obs=9,
138
+ weather_codes=999999999, precipitable_water=999, aerosol=0.999, snow_depth=999, days_since_snow=99,
139
+ albedo=999, precip_depth=999, precip_rate=99, horizontal_ir=9999, sky_cover=99,
140
+ )
141
+
142
+ #: Data source and uncertainty flags written to every row.
143
+ _DATA_FLAGS = "?9?9?9?9E0?9?9?9?9?9?9?9?9?9?9?9?9?9?9?9*9*9?9?9?9"
144
+
145
+
146
+ def describe_columns() -> pd.DataFrame:
147
+ """Return the input column contract as a table (name, unit, aggregation, required)."""
148
+ rows = [
149
+ dict(column=name, unit=spec["unit"], aggregation=spec["agg"], required=spec["required"],
150
+ description=spec["description"])
151
+ for name, spec in COLUMN_SPEC.items()
152
+ ]
153
+ return pd.DataFrame(rows).set_index("column")
154
+
155
+
156
+ @dataclass
157
+ class AmyReport:
158
+ """What was done to build an AMY: filled gaps, methods, caveats."""
159
+
160
+ year: int = 0
161
+ source_step_minutes: int = 60
162
+ filled_hours: Dict[str, int] = field(default_factory=dict)
163
+ decomposition: Dict[str, int] = field(default_factory=dict)
164
+ dni_capped_hours: int = 0
165
+ sky_cover_method: str = ""
166
+ sky_cover_calibration: Dict[str, float] = field(default_factory=dict)
167
+ notes: List[str] = field(default_factory=list)
168
+
169
+ def to_text(self) -> str:
170
+ lines = [f"AMY {self.year}: source resolution {self.source_step_minutes} min"]
171
+ if self.filled_hours:
172
+ lines.append("filled hours: " + ", ".join(f"{k}={v}" for k, v in sorted(self.filled_hours.items())))
173
+ if self.decomposition:
174
+ lines.append("diffuse/direct split (hours): " + ", ".join(f"{k}={v}" for k, v in self.decomposition.items()))
175
+ if self.sky_cover_method:
176
+ lines.append(f"sky cover: {self.sky_cover_method}")
177
+ lines.extend(self.notes)
178
+ return "\n".join(lines)
179
+
180
+
181
+ # --------------------------------------------------------------------------- input handling
182
+
183
+
184
+ def _validate_table(table: pd.DataFrame) -> pd.DataFrame:
185
+ if not isinstance(table.index, pd.DatetimeIndex):
186
+ raise ValueError("table needs a DatetimeIndex (see the module docstring for the datetime convention)")
187
+ t = table.copy()
188
+ if t.index.tz is not None:
189
+ t.index = t.index.tz_convert("UTC").tz_localize(None)
190
+ logger.warning("tz-aware index converted to UTC; pass table_utc_offset=0")
191
+ t = t[~t.index.isna()].sort_index()
192
+ if t.index.has_duplicates:
193
+ logger.warning("duplicate timestamps: keeping the first of each")
194
+ t = t[~t.index.duplicated(keep="first")]
195
+ known = [c for c in t.columns if c in COLUMN_SPEC]
196
+ ignored = [c for c in t.columns if c not in COLUMN_SPEC]
197
+ if ignored:
198
+ logger.info("ignoring unknown columns: %s", ", ".join(map(str, ignored)))
199
+ t = t[known].apply(pd.to_numeric, errors="coerce")
200
+ t = t.loc[:, t.notna().any()]
201
+ for name in ("temp_C", "ghi_Wm2", "wind_speed_ms", "wind_dir_deg"):
202
+ if name not in t.columns:
203
+ raise ValueError(f"required column '{name}' is missing or empty")
204
+ if "dewpoint_C" not in t.columns and "rh_pct" not in t.columns:
205
+ raise ValueError("one of 'dewpoint_C' or 'rh_pct' is required")
206
+ return t
207
+
208
+
209
+ def _infer_step_minutes(index: pd.DatetimeIndex) -> int:
210
+ diffs = pd.Series(index[1:] - index[:-1])
211
+ step = int(round(diffs.mode().iloc[0].total_seconds() / 60))
212
+ if step <= 0 or step > 60 or 60 % step != 0:
213
+ raise ValueError(f"unsupported time step of {step} minutes; use 1, 5, 10, 15, 30 or 60")
214
+ return step
215
+
216
+
217
+ def _vector_mean_direction(speed: Optional[pd.Series], direction: pd.Series, k: int) -> pd.Series:
218
+ """Speed-weighted vector mean of a wind direction over *k* sub-intervals."""
219
+ rad = np.radians(direction)
220
+ weight = speed if speed is not None else pd.Series(1.0, index=direction.index)
221
+ calm = weight == 0
222
+ u = (weight * np.sin(rad)).where(~calm, 0.0).where(direction.notna() | calm)
223
+ v = (weight * np.cos(rad)).where(~calm, 0.0).where(direction.notna() | calm)
224
+ mu = u.rolling(k, min_periods=k).mean()
225
+ mv = v.rolling(k, min_periods=k).mean()
226
+ out = np.degrees(np.arctan2(mu, mv)) % 360.0
227
+ out = out.where(np.hypot(mu, mv) > 1e-9, 0.0)
228
+ return out.where(mu.notna() & mv.notna())
229
+
230
+
231
+ def _subhourly_kt_std(ghi: pd.Series, location: dict, step: int) -> pd.Series:
232
+ """Standard deviation of the clearness index across the sub-intervals of each hour.
233
+
234
+ *ghi* is indexed by the interval end in local standard time. The result is
235
+ indexed the same way and is NaN where any sub-interval is missing or the
236
+ sun is low (zenith 85 degrees or more).
237
+ """
238
+ k = 60 // step
239
+ solar = _solar_frame(ghi.index - pd.Timedelta(minutes=step), location, interval_min=step)
240
+ kt = (ghi.clip(lower=0).to_numpy() / (solar["dni_extra"].to_numpy() * solar["cosz"].to_numpy()))
241
+ kt = pd.Series(kt, index=ghi.index).where(solar["zenith"].to_numpy() < _MAX_DECOMPOSE_ZENITH)
242
+ return kt.rolling(k, min_periods=k).std()
243
+
244
+
245
+ def aggregate_to_hourly(
246
+ table: pd.DataFrame,
247
+ location: dict,
248
+ *,
249
+ timestamp_label: str = "end",
250
+ table_utc_offset: float = 0.0,
251
+ ) -> Tuple[pd.DataFrame, int]:
252
+ """Aggregate a station table to hourly rows labelled by the interval *start* in local standard time.
253
+
254
+ Returns the hourly frame and the source time step in minutes. A column
255
+ ``ghi_kt_std`` (sub-hourly clearness variability) is added when the source
256
+ is finer than an hour and has global irradiance.
257
+ """
258
+ if timestamp_label not in ("end", "start"):
259
+ raise ValueError("timestamp_label must be 'end' or 'start'")
260
+ t = _validate_table(table)
261
+ step = _infer_step_minutes(t.index) if len(t) > 1 else 60
262
+ shift = float(location["utc_offset"]) - float(table_utc_offset)
263
+ if abs(shift - round(shift)) > 1e-9:
264
+ raise ValueError(
265
+ "the table's clock and the EPW time zone differ by a fractional number of hours; "
266
+ "convert the table to local standard time first"
267
+ )
268
+ idx = t.index
269
+ if timestamp_label == "start":
270
+ idx = idx + pd.Timedelta(minutes=step)
271
+ t.index = idx + pd.Timedelta(hours=int(round(shift)))
272
+ grid = pd.date_range(t.index.min(), t.index.max(), freq=f"{step}min")
273
+ t = t.reindex(grid)
274
+ if t.index[0].minute != 0 and step == 60:
275
+ raise ValueError("hourly timestamps must fall on the hour")
276
+ k = 60 // step
277
+
278
+ if k == 1:
279
+ hourly = t.copy()
280
+ else:
281
+ parts = {}
282
+ speed = t["wind_speed_ms"] if "wind_speed_ms" in t.columns else None
283
+ for name in t.columns:
284
+ how = COLUMN_SPEC[name]["agg"]
285
+ if how == "mean":
286
+ parts[name] = t[name].rolling(k, min_periods=k).mean()
287
+ elif how == "sum":
288
+ parts[name] = t[name].rolling(k, min_periods=k).sum()
289
+ elif how == "vector":
290
+ parts[name] = _vector_mean_direction(speed, t[name], k)
291
+ else:
292
+ parts[name] = t[name]
293
+ hourly = pd.DataFrame(parts)
294
+ if "ghi_Wm2" in t.columns:
295
+ hourly["ghi_kt_std"] = _subhourly_kt_std(t["ghi_Wm2"], location, step)
296
+ hourly = hourly[hourly.index.minute == 0]
297
+
298
+ hourly.index = hourly.index - pd.Timedelta(hours=1)
299
+ return hourly, step
300
+
301
+
302
+ # --------------------------------------------------------------------------- solar helpers
303
+
304
+
305
+ def _solar_frame(index_start: pd.DatetimeIndex, location: dict, interval_min: int = 60) -> pd.DataFrame:
306
+ """Solar position and extraterrestrial radiation at the middle of each interval.
307
+
308
+ *index_start* holds interval starts as tz-naive local standard time.
309
+ """
310
+ tz = _dt.timezone(_dt.timedelta(hours=float(location["utc_offset"])))
311
+ mid = (index_start + pd.Timedelta(minutes=interval_min / 2.0)).tz_localize(tz)
312
+ pos = pvlib.solarposition.get_solarposition(
313
+ mid, float(location["latitude"]), float(location["longitude"]), altitude=float(location["elevation"])
314
+ )
315
+ dni_extra = np.asarray(pvlib.irradiance.get_extra_radiation(mid), dtype=float)
316
+ zen = pos["zenith"].to_numpy()
317
+ cosz = np.cos(np.radians(np.minimum(zen, 89.0)))
318
+ return pd.DataFrame(
319
+ {
320
+ "zenith": zen,
321
+ "apparent_zenith": pos["apparent_zenith"].to_numpy(),
322
+ "cosz": cosz,
323
+ "dni_extra": dni_extra,
324
+ "ext_hor": dni_extra * np.maximum(np.cos(np.radians(zen)), 0.0),
325
+ },
326
+ index=index_start,
327
+ )
328
+
329
+
330
+ def decomposition_features(
331
+ ghi: pd.Series, sunshine_min: Optional[pd.Series], kt_std: Optional[pd.Series], solar: pd.DataFrame
332
+ ) -> pd.DataFrame:
333
+ """Hourly features of the diffuse-fraction tables (also used to fit them).
334
+
335
+ Columns ``kt`` (clearness index), ``S`` (sunshine fraction of the hour),
336
+ ``sd`` (sub-hourly clearness standard deviation) and ``zen`` (zenith, degrees).
337
+ """
338
+ kt = (ghi.clip(lower=0) / (solar["dni_extra"] * solar["cosz"])).clip(0.0, 1.2)
339
+ feats = pd.DataFrame({"kt": kt, "zen": solar["zenith"]}, index=ghi.index)
340
+ feats["S"] = (sunshine_min / 60.0).clip(0.0, 1.0) if sunshine_min is not None else np.nan
341
+ feats["sd"] = kt_std.clip(0.0, 0.5) if kt_std is not None else np.nan
342
+ return feats[["kt", "S", "sd", "zen"]]
343
+
344
+
345
+ @lru_cache(maxsize=1)
346
+ def _load_diffuse_tables() -> Dict[str, Tuple[List[str], list, np.ndarray]]:
347
+ resource = files("pyepwmorph.data").joinpath("amy_diffuse_table.parquet")
348
+ with resource.open("rb") as handle:
349
+ raw = pd.read_parquet(handle)
350
+ tables = {}
351
+ for name, frame in raw.groupby("table"):
352
+ axes_names = ["kt", "S", "sd", "zen"] if name == "kssd" else ["kt", "S", "zen"]
353
+ axes = [np.sort(frame[a].unique()) for a in axes_names]
354
+ frame = frame.sort_values(axes_names)
355
+ values = frame["kd"].to_numpy().reshape([len(a) for a in axes])
356
+ tables[name] = (axes_names, axes, values)
357
+ return tables
358
+
359
+
360
+ def _table_diffuse_fraction(feats: pd.DataFrame, name: str) -> pd.Series:
361
+ from scipy.interpolate import RegularGridInterpolator
362
+
363
+ axes_names, axes, values = _load_diffuse_tables()[name]
364
+ interp = RegularGridInterpolator(axes, values, bounds_error=False, fill_value=None)
365
+ pts = np.column_stack([
366
+ feats[a].clip(lower=ax[0], upper=ax[-1]).to_numpy() for a, ax in zip(axes_names, axes)
367
+ ])
368
+ return pd.Series(np.clip(interp(pts), 0.0, 1.0), index=feats.index)
369
+
370
+
371
+ def _decompose(
372
+ ghi: pd.Series,
373
+ hourly: pd.DataFrame,
374
+ solar: pd.DataFrame,
375
+ pressure_pa: pd.Series,
376
+ step: int,
377
+ method: str,
378
+ report: AmyReport,
379
+ ) -> Tuple[pd.Series, pd.Series]:
380
+ """Return diffuse horizontal and direct normal irradiance (W/m2)."""
381
+ if method not in ("auto", "table", "dirint", "erbs"):
382
+ raise ValueError("decomposition must be 'auto', 'table', 'dirint' or 'erbs'")
383
+ n = len(ghi)
384
+ dhi = pd.Series(np.nan, index=ghi.index)
385
+ how = pd.Series("", index=ghi.index, dtype=object)
386
+
387
+ daylight = (solar["zenith"] < _MAX_DECOMPOSE_ZENITH) & (ghi > 10.0)
388
+ dark = ~daylight
389
+ dhi[dark] = ghi[dark]
390
+ how[dark] = "night"
391
+
392
+ # measured components first
393
+ cosz = solar["cosz"]
394
+ if "dhi_Wm2" in hourly.columns:
395
+ m = hourly["dhi_Wm2"].notna() & daylight
396
+ dhi[m] = hourly["dhi_Wm2"][m].clip(lower=0).clip(upper=ghi[m])
397
+ how[m] = "measured"
398
+ if "dni_Wm2" in hourly.columns:
399
+ m = hourly["dni_Wm2"].notna() & daylight & dhi.isna()
400
+ dhi[m] = (ghi[m] - hourly["dni_Wm2"][m] * cosz[m]).clip(lower=0).clip(upper=ghi[m])
401
+ how[m] = "measured"
402
+
403
+ todo = dhi.isna() & daylight
404
+ sunshine = hourly["sunshine_min"] if "sunshine_min" in hourly.columns else None
405
+ kt_std = hourly["ghi_kt_std"] if "ghi_kt_std" in hourly.columns else None
406
+ feats = decomposition_features(ghi, sunshine, kt_std, solar)
407
+
408
+ use_tables = method in ("auto", "table") and sunshine is not None
409
+ if method == "table" and sunshine is None:
410
+ raise ValueError("decomposition='table' needs the 'sunshine_min' column")
411
+ if use_tables:
412
+ has_s = feats["S"].notna()
413
+ if step == 10:
414
+ m = todo & has_s & feats["sd"].notna()
415
+ if m.any():
416
+ dhi[m] = _table_diffuse_fraction(feats[m], "kssd") * ghi[m]
417
+ how[m] = "table (sunshine, variability)"
418
+ m = dhi.isna() & todo & has_s
419
+ if m.any():
420
+ dhi[m] = _table_diffuse_fraction(feats[m], "ks") * ghi[m]
421
+ how[m] = "table (sunshine)"
422
+
423
+ todo = dhi.isna() & daylight
424
+ if todo.any():
425
+ times = ghi.index + pd.Timedelta(minutes=30)
426
+ label = "erbs" if method == "erbs" else "dirint"
427
+ if label == "dirint":
428
+ tz = _dt.timezone(_dt.timedelta(hours=0))
429
+ dni = pvlib.irradiance.dirint(
430
+ ghi.to_numpy(), solar["zenith"].to_numpy(), times.tz_localize(tz), pressure=pressure_pa.to_numpy()
431
+ )
432
+ dni = pd.Series(np.asarray(dni, dtype=float), index=ghi.index)
433
+ fallback = (ghi - dni * cosz).clip(lower=0)
434
+ ok = todo & dni.notna()
435
+ dhi[ok] = fallback[ok]
436
+ how[ok] = "dirint"
437
+ todo = dhi.isna() & daylight
438
+ if todo.any():
439
+ er = pvlib.irradiance.erbs(ghi[todo], solar["zenith"][todo], pd.DatetimeIndex(ghi.index[todo]))
440
+ dhi[todo] = er["dhi"].to_numpy()
441
+ how[todo] = "erbs"
442
+
443
+ dhi = dhi.clip(lower=0.0)
444
+ dhi = pd.Series(np.minimum(dhi.to_numpy(), ghi.to_numpy()), index=ghi.index)
445
+ low_sun = solar["zenith"] >= _MAX_DECOMPOSE_ZENITH
446
+ dni = pd.Series(0.0, index=ghi.index)
447
+ ok = ~low_sun & (ghi > 0)
448
+ dni[ok] = (ghi[ok] - dhi[ok]) / cosz[ok]
449
+ cap = _MAX_DNI_FRACTION * solar["dni_extra"]
450
+ capped = dni > cap
451
+ report.dni_capped_hours = int(capped.sum())
452
+ if capped.any():
453
+ dni[capped] = cap[capped]
454
+ dhi[capped] = ghi[capped] - dni[capped] * cosz[capped]
455
+ dni = dni.clip(lower=0.0)
456
+ report.decomposition = {k: int(v) for k, v in how.value_counts().items() if k}
457
+ assert n == len(dni)
458
+ return dhi, dni
459
+
460
+
461
+ # --------------------------------------------------------------------------- sky cover
462
+
463
+
464
+ def estimate_sky_cover(
465
+ ghi: pd.Series,
466
+ solar: pd.DataFrame,
467
+ temp_c: pd.Series,
468
+ dewpoint_c: pd.Series,
469
+ lw_down: Optional[pd.Series],
470
+ location: dict,
471
+ report: AmyReport,
472
+ ) -> pd.Series:
473
+ """Total sky cover as a 0-1 fraction.
474
+
475
+ Daylight (zenith below 75 degrees): Kasten and Czeplak (1980) from the
476
+ ratio of measured to Ineichen clear-sky irradiance. Otherwise, from the
477
+ effective sky emissivity of the downwelling longwave: clear-sky emissivity
478
+ from Martin and Berdahl (1984), cloud fraction as the share of the way to
479
+ an overcast sky, then a straight-line fit to the daylight estimate removes
480
+ the site's bias. Without longwave data, night values are carried across
481
+ from daylight by interpolation.
482
+ """
483
+ mid = (ghi.index + pd.Timedelta(minutes=30)).tz_localize(_dt.timezone(_dt.timedelta(hours=float(location["utc_offset"]))))
484
+ try:
485
+ linke = pvlib.clearsky.lookup_linke_turbidity(mid, float(location["latitude"]), float(location["longitude"]))
486
+ linke = np.asarray(linke, dtype=float)
487
+ except Exception as exc: # the lookup needs pvlib's bundled turbidity file
488
+ logger.warning("Linke turbidity lookup failed (%s); using 3.0", exc)
489
+ linke = np.full(len(mid), 3.0)
490
+ loc = pvlib.location.Location(float(location["latitude"]), float(location["longitude"]),
491
+ altitude=float(location["elevation"]))
492
+ clear = loc.get_clearsky(mid, model="ineichen", linke_turbidity=linke)["ghi"].to_numpy()
493
+ day = (solar["zenith"] < 75.0).to_numpy()
494
+ kc = np.where(day & (clear > 50), ghi.clip(lower=0).to_numpy() / np.where(clear > 50, clear, np.nan), np.nan)
495
+ n_ghi = pd.Series(((1.0 - np.clip(kc, None, 1.0)) / 0.75).clip(0, 1) ** (1 / 3.4), index=ghi.index)
496
+
497
+ n_lw = None
498
+ if lw_down is not None:
499
+ sigma = 5.670374419e-8
500
+ eps = lw_down / (sigma * (temp_c + 273.15) ** 4)
501
+ solar_hour = ((mid.hour + 0.5).to_numpy() + (float(location["longitude"]) - 15.0 * float(location["utc_offset"])) / 15.0)
502
+ eps_clear = 0.711 + 0.0056 * dewpoint_c + 0.000073 * dewpoint_c ** 2 + 0.013 * np.cos(2 * np.pi * solar_hour / 24.0)
503
+ n_lw = ((eps - eps_clear) / (1.0 - eps_clear)).clip(0, 1)
504
+
505
+ cover = n_ghi.copy()
506
+ if n_lw is not None:
507
+ pair = n_ghi.notna() & n_lw.notna() & (solar["zenith"] < 70.0)
508
+ method = "longwave"
509
+ if pair.sum() >= 200 and n_ghi[pair].corr(n_lw[pair]) > 0.5:
510
+ slope, intercept = np.polyfit(n_lw[pair], n_ghi[pair], 1)
511
+ if 0.5 <= slope <= 3.0:
512
+ n_lw = (intercept + slope * n_lw).clip(0, 1)
513
+ method = "longwave calibrated to daylight clear-sky index"
514
+ report.sky_cover_calibration = dict(
515
+ slope=float(slope), intercept=float(intercept),
516
+ correlation=float(n_ghi[pair].corr(n_lw[pair])), hours=int(pair.sum()),
517
+ )
518
+ cover = n_ghi.where(n_ghi.notna(), n_lw)
519
+ report.sky_cover_method = f"daylight: clear-sky index (Kasten & Czeplak 1980); otherwise {method} (Martin & Berdahl 1984)"
520
+ else:
521
+ cover = n_ghi.interpolate(limit=24, limit_direction="both")
522
+ report.sky_cover_method = "daylight: clear-sky index (Kasten & Czeplak 1980); night interpolated, no longwave data"
523
+ return cover
524
+
525
+
526
+ # --------------------------------------------------------------------------- gaps
527
+
528
+
529
+ def _runs(mask: pd.Series) -> List[Tuple[pd.Timestamp, int]]:
530
+ """Start and length of each run of True in *mask*."""
531
+ out = []
532
+ values = mask.to_numpy()
533
+ i = 0
534
+ while i < len(values):
535
+ if values[i]:
536
+ j = i
537
+ while j < len(values) and values[j]:
538
+ j += 1
539
+ out.append((mask.index[i], j - i))
540
+ i = j
541
+ else:
542
+ i += 1
543
+ return out
544
+
545
+
546
+ def _fill_gaps(
547
+ hourly: pd.DataFrame, solar: pd.DataFrame, max_gap_hours: int, long_gap: str, report: AmyReport
548
+ ) -> pd.DataFrame:
549
+ if long_gap not in ("raise", "interpolate"):
550
+ raise ValueError("long_gap must be 'raise' or 'interpolate'")
551
+ h = hourly.copy()
552
+ limit = None if long_gap == "interpolate" else int(max_gap_hours)
553
+
554
+ problems = []
555
+ for name in _REQUIRED_FOR_GAPS:
556
+ if name in h.columns:
557
+ for start, length in _runs(h[name].isna()):
558
+ if length > max_gap_hours:
559
+ problems.append(f"{name}: {length} h from {start:%Y-%m-%d %H:%M}")
560
+ if problems and long_gap == "raise":
561
+ raise ValueError(
562
+ f"gaps longer than {max_gap_hours} h: " + "; ".join(problems[:10])
563
+ + ". Fill them, or pass long_gap='interpolate'."
564
+ )
565
+ if problems:
566
+ report.notes.append("long gaps interpolated: " + "; ".join(problems[:10]))
567
+
568
+ for name in _LINEAR_COLUMNS:
569
+ if name in h.columns:
570
+ before = int(h[name].isna().sum())
571
+ h[name] = h[name].interpolate(method="linear", limit=limit, limit_direction="both")
572
+ after = int(h[name].isna().sum())
573
+ if before - after:
574
+ report.filled_hours[name] = before - after
575
+
576
+ if "wind_dir_deg" in h.columns:
577
+ rad = np.radians(h["wind_dir_deg"])
578
+ u, v = np.sin(rad), np.cos(rad)
579
+ before = int(h["wind_dir_deg"].isna().sum())
580
+ u = u.interpolate(limit=limit, limit_direction="both")
581
+ v = v.interpolate(limit=limit, limit_direction="both")
582
+ h["wind_dir_deg"] = (np.degrees(np.arctan2(u, v)) % 360.0).where(u.notna() & v.notna())
583
+ if before - int(h["wind_dir_deg"].isna().sum()):
584
+ report.filled_hours["wind_dir_deg"] = before - int(h["wind_dir_deg"].isna().sum())
585
+
586
+ # irradiance: interpolate the clearness index so a gap keeps the sun's shape
587
+ ghi = h["ghi_Wm2"].clip(lower=0)
588
+ denom = solar["dni_extra"] * solar["cosz"]
589
+ high_sun = solar["zenith"] < _MAX_DECOMPOSE_ZENITH
590
+ before = int(ghi.isna().sum())
591
+ kt_filled = (ghi / denom).where(high_sun).interpolate(limit=limit, limit_direction="both")
592
+ ghi = (kt_filled * denom).where(high_sun, ghi.interpolate(limit=limit, limit_direction="both"))
593
+ # an hour is dark when the sun stays below the horizon for all of it (the sun moves at most 8 degrees)
594
+ ghi = ghi.where(solar["zenith"] < _DARK_ZENITH, 0.0)
595
+ h["ghi_Wm2"] = ghi
596
+ if before - int(ghi.isna().sum()):
597
+ report.filled_hours["ghi_Wm2"] = before - int(ghi.isna().sum())
598
+ return h
599
+
600
+
601
+ # --------------------------------------------------------------------------- build
602
+
603
+
604
+ def build_amy_dataframe(
605
+ table: pd.DataFrame,
606
+ location: dict,
607
+ year: int,
608
+ *,
609
+ timestamp_label: str = "end",
610
+ table_utc_offset: float = 0.0,
611
+ max_gap_hours: int = 6,
612
+ long_gap: str = "raise",
613
+ decomposition: str = "auto",
614
+ ) -> Tuple[pd.DataFrame, AmyReport]:
615
+ """Build the 8760 EPW data rows for *year* from a station table.
616
+
617
+ Parameters
618
+ ----------
619
+ table : pd.DataFrame
620
+ Station measurements; see the module docstring for columns and units.
621
+ location : dict
622
+ ``latitude``, ``longitude``, ``elevation`` (m) and ``utc_offset``
623
+ (hours east of UTC, standard time), plus optional ``site``, ``province``,
624
+ ``country_code``, ``type`` and ``usaf`` for the EPW header.
625
+ year : int
626
+ The calendar year, in local standard time.
627
+ timestamp_label : {"end", "start"}
628
+ Which end of each averaging interval the index marks.
629
+ table_utc_offset : float
630
+ Hours east of UTC of the table's clock (0 for UTC).
631
+ max_gap_hours : int
632
+ Longest gap that is interpolated.
633
+ long_gap : {"raise", "interpolate"}
634
+ What to do about longer gaps in the required columns.
635
+ decomposition : {"auto", "table", "dirint", "erbs"}
636
+ How to split global irradiance when no components are measured.
637
+
638
+ Returns
639
+ -------
640
+ tuple
641
+ A data frame with the 35 EPW columns (``hour`` runs 1-24) and an :class:`AmyReport`.
642
+ """
643
+ for key in ("latitude", "longitude", "elevation", "utc_offset"):
644
+ if key not in location:
645
+ raise ValueError(f"location is missing '{key}'")
646
+ report = AmyReport(year=int(year))
647
+ hourly, step = aggregate_to_hourly(table, location, timestamp_label=timestamp_label, table_utc_offset=table_utc_offset)
648
+ report.source_step_minutes = step
649
+
650
+ full = pd.date_range(f"{year}-01-01 00:00", f"{year}-12-31 23:00", freq="h")
651
+ present = hourly.index.intersection(full)
652
+ if len(present) < 0.5 * len(full):
653
+ raise ValueError(f"the table covers only {len(present)} of the {len(full)} hours of {year}")
654
+ hourly = hourly.reindex(full)
655
+ leap_day = (full.month == 2) & (full.day == 29)
656
+ if leap_day.any():
657
+ hourly = hourly[~leap_day]
658
+ report.notes.append("29 February dropped")
659
+
660
+ solar = _solar_frame(hourly.index, location)
661
+ hourly = _fill_gaps(hourly, solar, max_gap_hours, long_gap, report)
662
+
663
+ for name in ("temp_C", "ghi_Wm2", "wind_speed_ms", "wind_dir_deg"):
664
+ left = int(hourly[name].isna().sum())
665
+ if left:
666
+ raise ValueError(f"{left} hours of '{name}' are still missing after filling")
667
+
668
+ temp = hourly["temp_C"]
669
+ if "dewpoint_C" in hourly.columns and "rh_pct" in hourly.columns:
670
+ dew, rh = hourly["dewpoint_C"], hourly["rh_pct"]
671
+ elif "dewpoint_C" in hourly.columns:
672
+ dew = hourly["dewpoint_C"]
673
+ rh = 100.0 * pd.Series(
674
+ psychrometrics.saturated_vapor_pressure(dew + 273.15) / psychrometrics.saturated_vapor_pressure(temp + 273.15),
675
+ index=hourly.index,
676
+ )
677
+ else:
678
+ rh = hourly["rh_pct"].clip(0, 100)
679
+ dew = pd.Series(psychrometrics.dew_point_from_db_rh(temp.to_numpy(), rh.to_numpy()), index=hourly.index)
680
+ if dew.isna().any() or rh.isna().any():
681
+ raise ValueError("humidity is still missing after filling")
682
+ rh = rh.clip(0, 100)
683
+ dew = np.minimum(dew, temp)
684
+
685
+ if "pressure_Pa" in hourly.columns and hourly["pressure_Pa"].notna().all():
686
+ pressure = hourly["pressure_Pa"]
687
+ else:
688
+ std = float(pvlib.atmosphere.alt2pres(float(location["elevation"])))
689
+ pressure = pd.Series(std, index=hourly.index)
690
+ report.notes.append("pressure from the standard atmosphere at the station elevation")
691
+
692
+ ghi = hourly["ghi_Wm2"].clip(lower=0)
693
+ dhi, dni = _decompose(ghi, hourly, solar, pressure, step, decomposition, report)
694
+
695
+ lw = hourly["lw_down_Wm2"] if "lw_down_Wm2" in hourly.columns else None
696
+ if "total_sky_cover_tenths" in hourly.columns and hourly["total_sky_cover_tenths"].notna().any():
697
+ cover = (hourly["total_sky_cover_tenths"] / 10.0).clip(0, 1)
698
+ report.sky_cover_method = "observed"
699
+ else:
700
+ cover = estimate_sky_cover(ghi, solar, temp, dew, lw, location, report)
701
+ tenths = np.rint(cover * 10.0)
702
+ total = tenths.where(tenths.notna(), _MISSING["sky_cover"]).astype(int)
703
+ if "opaque_sky_cover_tenths" in hourly.columns and hourly["opaque_sky_cover_tenths"].notna().all():
704
+ opaque = hourly["opaque_sky_cover_tenths"].round().astype(int)
705
+ else:
706
+ opaque = total
707
+
708
+ start = hourly.index
709
+ n = len(hourly)
710
+
711
+ def col(name, default):
712
+ if name in hourly.columns:
713
+ return hourly[name].fillna(default)
714
+ return pd.Series(default, index=start)
715
+
716
+ wind_speed = hourly["wind_speed_ms"].clip(lower=0)
717
+ wind_dir = hourly["wind_dir_deg"].where(wind_speed > 0, 0.0).round()
718
+
719
+ epw = pd.DataFrame(index=start, columns=EPW_COLUMN_NAMES, dtype=object)
720
+ epw["year"] = int(year)
721
+ epw["month"] = start.month
722
+ epw["day"] = start.day
723
+ epw["hour"] = start.hour + 1
724
+ epw["minute"] = 0
725
+ epw["datasource"] = _DATA_FLAGS
726
+ epw["drybulb_C"] = temp.round(1)
727
+ epw["dewpoint_C"] = dew.round(1)
728
+ epw["relhum_percent"] = rh.round(1)
729
+ epw["atmos_Pa"] = pressure.round(0).astype(int)
730
+ epw["exthorrad_Whm2"] = solar["ext_hor"].round(0).astype(int)
731
+ epw["extdirrad_Whm2"] = solar["dni_extra"].round(0).astype(int)
732
+ epw["horirsky_Whm2"] = (
733
+ lw.round(0).astype(int) if lw is not None and lw.notna().all() else _MISSING["horizontal_ir"]
734
+ )
735
+ epw["glohorrad_Whm2"] = ghi.round(0).astype(int)
736
+ epw["dirnorrad_Whm2"] = dni.round(0).astype(int)
737
+ epw["difhorrad_Whm2"] = dhi.round(0).astype(int)
738
+ epw["glohorillum_lux"] = _MISSING["illuminance"]
739
+ epw["dirnorillum_lux"] = _MISSING["illuminance"]
740
+ epw["difhorillum_lux"] = _MISSING["illuminance"]
741
+ epw["zenlum_lux"] = _MISSING["zenith_luminance"]
742
+ epw["winddir_deg"] = wind_dir.astype(int)
743
+ epw["windspd_ms"] = wind_speed.round(1)
744
+ epw["totskycvr_tenths"] = total
745
+ epw["opaqskycvr_tenths"] = opaque
746
+ epw["visibility_km"] = (
747
+ col("visibility_km", _MISSING["visibility"]).round(1) if "visibility_km" in hourly.columns else _MISSING["visibility"]
748
+ )
749
+ epw["ceiling_hgt_m"] = (
750
+ col("ceiling_height_m", _MISSING["ceiling"]).round(0).astype(int)
751
+ if "ceiling_height_m" in hourly.columns else _MISSING["ceiling"]
752
+ )
753
+ epw["presweathobs"] = _MISSING["weather_obs"]
754
+ epw["presweathcodes"] = _MISSING["weather_codes"]
755
+ epw["precip_wtr_mm"] = _MISSING["precipitable_water"]
756
+ epw["aerosol_opt_thousandths"] = _MISSING["aerosol"]
757
+ epw["snowdepth_cm"] = (
758
+ col("snow_depth_cm", _MISSING["snow_depth"]).clip(lower=0).round(0).astype(int)
759
+ if "snow_depth_cm" in hourly.columns else _MISSING["snow_depth"]
760
+ )
761
+ epw["days_last_snow"] = _MISSING["days_since_snow"]
762
+ epw["Albedo"] = _MISSING["albedo"]
763
+ if "precip_mm" in hourly.columns:
764
+ precip = hourly["precip_mm"]
765
+ epw["liq_precip_depth_mm"] = precip.clip(lower=0).round(1).where(precip.notna(), _MISSING["precip_depth"])
766
+ epw["liq_precip_rate_Hour"] = np.where(precip.notna(), 1.0, _MISSING["precip_rate"])
767
+ else:
768
+ epw["liq_precip_depth_mm"] = _MISSING["precip_depth"]
769
+ epw["liq_precip_rate_Hour"] = _MISSING["precip_rate"]
770
+ assert len(epw) == n == 8760
771
+ return epw, report
772
+
773
+
774
+ # --------------------------------------------------------------------------- output
775
+
776
+ _FORMATS = {
777
+ "year": "{:d}", "month": "{:d}", "day": "{:d}", "hour": "{:d}", "minute": "{:d}",
778
+ "drybulb_C": "{:.1f}", "dewpoint_C": "{:.1f}", "relhum_percent": "{:.1f}", "atmos_Pa": "{:d}",
779
+ "windspd_ms": "{:.1f}", "visibility_km": "{:.1f}", "liq_precip_depth_mm": "{:.1f}",
780
+ "liq_precip_rate_Hour": "{:.1f}", "aerosol_opt_thousandths": "{:.3f}",
781
+ }
782
+
783
+
784
+ def _ascii(value) -> str:
785
+ """Plain-ASCII text for header fields (EnergyPlus is happiest without accents)."""
786
+ return unicodedata.normalize("NFKD", str(value)).encode("ascii", "ignore").decode("ascii")
787
+
788
+
789
+ def _format_rows(epw: pd.DataFrame) -> List[str]:
790
+ cols = []
791
+ for name in EPW_COLUMN_NAMES:
792
+ values = epw[name].tolist()
793
+ if name == "datasource":
794
+ cols.append([str(v) for v in values])
795
+ continue
796
+ fmt = _FORMATS.get(name, "{:d}")
797
+ cols.append([fmt.format(float(v) if "f}" in fmt else int(v)) for v in values])
798
+ return [",".join(row) + "\n" for row in zip(*cols)]
799
+
800
+
801
+ def write_amy_epw(
802
+ path: str,
803
+ epw: pd.DataFrame,
804
+ location: dict,
805
+ report: AmyReport,
806
+ *,
807
+ source_name: str = "weather station",
808
+ attribution: Optional[str] = None,
809
+ ) -> None:
810
+ """Write the data rows from :func:`build_amy_dataframe` to an EPW file.
811
+
812
+ The header states the period of record (``Period of Record=<year>-<year>``)
813
+ and the methods used for derived fields, so downstream tools that read the
814
+ baseline period see a single year.
815
+ """
816
+ year = int(epw["year"].iloc[0])
817
+ first = _dt.date(year, 1, 1)
818
+ day_of_week = first.strftime("%A")
819
+
820
+ def text(key, default):
821
+ return _ascii(location.get(key, default)).replace(",", " ")
822
+
823
+ loc_line = ",".join([
824
+ "LOCATION", text("site", "Unknown"), text("province", "-"), text("country_code", "-"),
825
+ text("type", "Measured"), text("usaf", "999999"),
826
+ f"{float(location['latitude']):.5f}", f"{float(location['longitude']):.5f}",
827
+ f"{float(location['utc_offset']):.1f}", f"{float(location['elevation']):.1f}",
828
+ ])
829
+
830
+ methods = []
831
+ if report.decomposition:
832
+ methods.append("direct/diffuse split: " + ", ".join(f"{k} {v} h" for k, v in report.decomposition.items()))
833
+ if report.sky_cover_method:
834
+ methods.append("sky cover: " + report.sky_cover_method)
835
+ if report.filled_hours:
836
+ methods.append("interpolated hours: " + ", ".join(f"{k} {v}" for k, v in sorted(report.filled_hours.items())))
837
+ comments_2 = (
838
+ f"Actual meteorological year {year}; hourly means in local standard time, hour ending; "
839
+ + "; ".join(methods)
840
+ + ("; " + attribution if attribution else "")
841
+ ).replace('"', "'")
842
+ comments_2 = _ascii(comments_2)
843
+ comments_1 = (
844
+ f"Measured data from {source_name}, built with pyepwmorph; Period of Record={year}-{year}"
845
+ ).replace('"', "'")
846
+ comments_1 = _ascii(comments_1)
847
+
848
+ header = [
849
+ loc_line,
850
+ "DESIGN CONDITIONS,0",
851
+ "TYPICAL/EXTREME PERIODS,0",
852
+ "GROUND TEMPERATURES,0",
853
+ "HOLIDAYS/DAYLIGHT SAVINGS,No,0,0,0",
854
+ f'COMMENTS 1,"{comments_1}"',
855
+ f'COMMENTS 2,"{comments_2}"',
856
+ f"DATA PERIODS,1,1,Data,{day_of_week},1/1,12/31",
857
+ ]
858
+ with open(path, "w", encoding="utf-8") as fh:
859
+ fh.write("\n".join(header) + "\n")
860
+ fh.writelines(_format_rows(epw))
861
+
862
+
863
+ def station_table_to_epw(
864
+ table: pd.DataFrame,
865
+ location: dict,
866
+ year: int,
867
+ output_path: str,
868
+ *,
869
+ source_name: str = "weather station",
870
+ attribution: Optional[str] = None,
871
+ **build_kwargs,
872
+ ) -> AmyReport:
873
+ """Build an AMY from a station table and write it to *output_path*.
874
+
875
+ Keyword arguments other than ``source_name`` and ``attribution`` go to
876
+ :func:`build_amy_dataframe`. Returns the :class:`AmyReport`.
877
+ """
878
+ epw, report = build_amy_dataframe(table, location, year, **build_kwargs)
879
+ write_amy_epw(output_path, epw, location, report, source_name=source_name, attribution=attribution)
880
+ return report
@@ -0,0 +1,135 @@
1
+ """MeteoSwiss open data (ogd-smn) adapter for :mod:`pyepwmorph.tools.amy`.
2
+
3
+ .. deprecated:: 3.4.0
4
+ Moved to ``weather_file_builder.amy_meteoswiss``. Removed in pyepwmorph 4.0.
5
+
6
+ Reads the automatic station files that MeteoSwiss publishes at
7
+ https://opendatadocs.meteoswiss.ch/ (10-minute ``..._t_...`` and hourly
8
+ ``..._h_...`` CSVs, semicolon separated) and renames their parameter codes to
9
+ the columns :func:`pyepwmorph.tools.amy.build_amy_dataframe` expects. The
10
+ station table is the ``ogd-smn_meta_stations.csv`` file.
11
+
12
+ MeteoSwiss timestamps (``reference_timestamp``, ``DD.MM.YYYY HH:MM``) are UTC
13
+ and mark the *end* of the averaging interval, which is what the defaults of
14
+ ``build_amy_dataframe`` assume.
15
+
16
+ Data: Source MeteoSwiss, open government data (check the current terms for
17
+ attribution wording).
18
+ """
19
+
20
+ import logging
21
+ import warnings as _warnings
22
+ from typing import Optional
23
+
24
+ import pandas as pd
25
+
26
+ _warnings.warn(
27
+ "pyepwmorph.tools.amy_meteoswiss is deprecated and will be removed in pyepwmorph 4.0. "
28
+ "It has moved to weather-file-builder (pip install 'weather-file-builder>=2.1'): "
29
+ "use weather_file_builder.amy_meteoswiss.",
30
+ DeprecationWarning,
31
+ stacklevel=2,
32
+ )
33
+
34
+ logger = logging.getLogger(__name__)
35
+
36
+ #: Attribution line for the EPW header.
37
+ ATTRIBUTION = "Source: MeteoSwiss (Federal Office of Meteorology and Climatology), open government data"
38
+
39
+ #: Parameter code -> (amy column, multiplier). 10-minute files (``z``/``s`` suffixes).
40
+ MAPPING_10MIN = {
41
+ "tre200s0": ("temp_C", 1.0),
42
+ "tde200s0": ("dewpoint_C", 1.0),
43
+ "ure200s0": ("rh_pct", 1.0),
44
+ "prestas0": ("pressure_Pa", 100.0),
45
+ "fkl010z0": ("wind_speed_ms", 1.0),
46
+ "dkl010z0": ("wind_dir_deg", 1.0),
47
+ "gre000z0": ("ghi_Wm2", 1.0),
48
+ "ods000z0": ("dhi_Wm2", 1.0),
49
+ "oli000z0": ("lw_down_Wm2", 1.0),
50
+ "sre000z0": ("sunshine_min", 1.0),
51
+ "rre150z0": ("precip_mm", 1.0),
52
+ "htoauts0": ("snow_depth_cm", 1.0),
53
+ }
54
+
55
+ #: Hourly files (``h`` suffix).
56
+ MAPPING_HOURLY = {
57
+ "tre200h0": ("temp_C", 1.0),
58
+ "tde200h0": ("dewpoint_C", 1.0),
59
+ "ure200h0": ("rh_pct", 1.0),
60
+ "prestah0": ("pressure_Pa", 100.0),
61
+ "fkl010h0": ("wind_speed_ms", 1.0),
62
+ "dkl010h0": ("wind_dir_deg", 1.0),
63
+ "gre000h0": ("ghi_Wm2", 1.0),
64
+ "ods000h0": ("dhi_Wm2", 1.0),
65
+ "oli000h0": ("lw_down_Wm2", 1.0),
66
+ "sre000h0": ("sunshine_min", 1.0),
67
+ "rre150h0": ("precip_mm", 1.0),
68
+ "htoauths": ("snow_depth_cm", 1.0),
69
+ }
70
+
71
+ _TIMESTAMP_FORMAT = "%d.%m.%Y %H:%M"
72
+
73
+
74
+ def read_meteoswiss_ogd(path: str, station: Optional[str] = None) -> pd.DataFrame:
75
+ """Read a MeteoSwiss ogd-smn station CSV into the AMY input table.
76
+
77
+ Parameters
78
+ ----------
79
+ path : str
80
+ A 10-minute or hourly station file (decade files such as
81
+ ``ogd-smn_sma_t_historical_2020-2029.csv`` work; so do the ``recent`` files).
82
+ station : str, optional
83
+ Station abbreviation to keep if the file holds several.
84
+
85
+ Returns
86
+ -------
87
+ pd.DataFrame
88
+ Tz-naive UTC timestamps marking the end of each interval, columns named
89
+ as in :data:`pyepwmorph.tools.amy.COLUMN_SPEC` with units converted
90
+ (pressure in Pa).
91
+ """
92
+ raw = pd.read_csv(path, sep=";", encoding="latin-1")
93
+ if "reference_timestamp" not in raw.columns:
94
+ raise ValueError(f"{path} has no 'reference_timestamp' column; is it a MeteoSwiss ogd-smn file?")
95
+ if station is not None and "station_abbr" in raw.columns:
96
+ raw = raw[raw["station_abbr"] == station]
97
+ if any(c in raw.columns for c in MAPPING_10MIN):
98
+ mapping = MAPPING_10MIN
99
+ elif any(c in raw.columns for c in MAPPING_HOURLY):
100
+ mapping = MAPPING_HOURLY
101
+ else:
102
+ raise ValueError(f"{path} has none of the expected MeteoSwiss parameter codes")
103
+ index = pd.to_datetime(raw["reference_timestamp"], format=_TIMESTAMP_FORMAT)
104
+ out = pd.DataFrame(index=index)
105
+ for code, (name, factor) in mapping.items():
106
+ if code in raw.columns:
107
+ out[name] = pd.to_numeric(raw[code], errors="coerce").to_numpy() * factor
108
+ out.index.name = "timestamp_utc_end"
109
+ return out
110
+
111
+
112
+ def meteoswiss_location(stations_csv: str, station: str, utc_offset: float = 1.0) -> dict:
113
+ """EPW location for a station from ``ogd-smn_meta_stations.csv``.
114
+
115
+ ``utc_offset`` is the standard-time offset (1.0 for Switzerland; the data
116
+ stay in fixed standard time, without daylight saving).
117
+ """
118
+ meta = pd.read_csv(stations_csv, sep=";", encoding="latin-1")
119
+ row = meta[meta["station_abbr"] == station]
120
+ if row.empty:
121
+ raise ValueError(f"station '{station}' not found in {stations_csv}")
122
+ row = row.iloc[0]
123
+ wigos = str(row.get("station_wigos_id", ""))
124
+ wmo = wigos.split("-")[-1] if wigos else "999999"
125
+ return dict(
126
+ site=str(row["station_name"]).replace(" / ", "-"),
127
+ province=str(row["station_canton"]),
128
+ country_code="CHE",
129
+ type=f"MeteoSwiss-{station}",
130
+ usaf=wmo,
131
+ latitude=float(row["station_coordinates_wgs84_lat"]),
132
+ longitude=float(row["station_coordinates_wgs84_lon"]),
133
+ utc_offset=float(utc_offset),
134
+ elevation=float(row["station_height_masl"]),
135
+ )
@@ -17,7 +17,7 @@ core-metadata-version = "2.4"
17
17
 
18
18
  [project]
19
19
  name = "pyepwmorph"
20
- version = "3.2.0"
20
+ version = "3.4.0"
21
21
  authors = [
22
22
  { name="Justin McCarty", email="mccarty.justin.f@gmail.com" },
23
23
  ]
File without changes