buildingdata 0.3.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {buildingdata-0.3.0 → buildingdata-0.5.0}/PKG-INFO +17 -10
  2. {buildingdata-0.3.0 → buildingdata-0.5.0}/README.md +16 -9
  3. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/__init__.py +1 -0
  4. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/gcs.py +16 -1
  5. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/reference/__init__.py +1 -0
  6. buildingdata-0.5.0/buildingdata/reference/diagnosis.py +108 -0
  7. buildingdata-0.5.0/buildingdata/reference/elecdom.py +153 -0
  8. buildingdata-0.5.0/buildingdata/reference/occupant_diaries.py +257 -0
  9. buildingdata-0.5.0/buildingdata/tests/test_gcs.py +73 -0
  10. buildingdata-0.5.0/buildingdata/tests/test_pipeline_diagnosis.py +283 -0
  11. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/test_public_api.py +1 -0
  12. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/test_reference.py +262 -22
  13. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/validation.py +52 -2
  14. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata.egg-info/PKG-INFO +17 -10
  15. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata.egg-info/SOURCES.txt +3 -0
  16. {buildingdata-0.3.0 → buildingdata-0.5.0}/pyproject.toml +1 -1
  17. buildingdata-0.3.0/buildingdata/reference/diagnosis.py +0 -74
  18. buildingdata-0.3.0/buildingdata/reference/occupant_diaries.py +0 -41
  19. {buildingdata-0.3.0 → buildingdata-0.5.0}/LICENSE +0 -0
  20. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/_cli.py +0 -0
  21. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/bulk.py +0 -0
  22. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/cache.py +0 -0
  23. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/config.py +0 -0
  24. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/exceptions.py +0 -0
  25. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/reference/census.py +0 -0
  26. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/reference/districts.py +0 -0
  27. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/reference/elmas.py +0 -0
  28. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/reference/enedis.py +0 -0
  29. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/reference/gas_network.py +0 -0
  30. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/reference/ore.py +0 -0
  31. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/simulation/__init__.py +0 -0
  32. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/simulation/_epw.py +0 -0
  33. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/simulation/bdtopo.py +0 -0
  34. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/simulation/bdtopo_bulk.py +0 -0
  35. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/simulation/era5.py +0 -0
  36. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/simulation/era5_bulk.py +0 -0
  37. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/__init__.py +0 -0
  38. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/conftest.py +0 -0
  39. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/test_bulk.py +0 -0
  40. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/test_cache.py +0 -0
  41. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/test_config.py +0 -0
  42. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata/tests/test_simulation.py +0 -0
  43. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata.egg-info/dependency_links.txt +0 -0
  44. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata.egg-info/entry_points.txt +0 -0
  45. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata.egg-info/requires.txt +0 -0
  46. {buildingdata-0.3.0 → buildingdata-0.5.0}/buildingdata.egg-info/top_level.txt +0 -0
  47. {buildingdata-0.3.0 → buildingdata-0.5.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: buildingdata
3
- Version: 0.3.0
3
+ Version: 0.5.0
4
4
  Summary: Data management layer for buildingmodel — reference data download, BDTOPO retrieval, ERA5 weather
5
5
  License-Expression: MIT
6
6
  Requires-Python: >=3.10
@@ -76,18 +76,25 @@ buildingdata configure --bucket my-bucket --cache-dir ~/.cache/buildingdata
76
76
 
77
77
  ## What it provides
78
78
 
79
- | Function | Source | Returns |
80
- | --- | --- | --- |
81
- | `get_census()` | INSEE census (GCS) | polars `DataFrame` |
82
- | `get_districts()` | IRIS geometries (GCS) | geopandas `GeoDataFrame` |
83
- | `get_diagnosis()` | ADEME energy performance diagnoses (GCS) | polars `DataFrame` |
84
- | `get_gas_network()` | GRDF gas network routes (GCS) | geopandas `GeoDataFrame` |
85
- | `get_bdtopo(iris_code)` | IGN Géoplateforme WFS (live) | geopandas `GeoDataFrame` |
86
- | `get_era5_climate(lat, lon, year)` | Copernicus CDS (live) | path to EPW file |
79
+ | Function | Source | Returns | Consumed by |
80
+ | --- | --- | --- | --- |
81
+ | `get_census()` | INSEE census (GCS) | polars `DataFrame` | buildingmodel — dwelling inference |
82
+ | `get_districts()` | IRIS geometries (GCS) | geopandas `GeoDataFrame` | buildingmodel — geocoding; building_eload — district lookup |
83
+ | `get_diagnosis()` | ADEME energy performance diagnoses (GCS) | polars `DataFrame` | buildingmodel — envelope & system inference |
84
+ | `get_gas_network()` | GRDF gas network routes (GCS) | geopandas `GeoDataFrame` | buildingmodel — gas-connection inference |
85
+ | `get_ore()` | Agence ORE annual consumption per IRIS (GCS) | polars `DataFrame` | building_eload — calibration target |
86
+ | `get_enedis_national()` / `get_enedis_regional()` | Enedis conso-inf36 measured consumption (GCS) | polars `DataFrame` | building_eload — validation reference |
87
+ | `get_elmas(table)` | ELMAS non-residential load curves (GCS) | polars `DataFrame` | building_eload — non-residential model |
88
+ | `get_occupant_diaries()` | synthetic occupant activity calendar (GCS) | polars `DataFrame` | building_eload — occupancy models |
89
+ | `get_elecdom()` | Enedis panel Elecdom end-use load curves (GCS) | polars `DataFrame` | provenance only — source of building_eload constants |
90
+ | `get_bdtopo(iris_code)` | IGN Géoplateforme WFS (live) or bulk cache | geopandas `GeoDataFrame` | building_eload → buildingmodel — input geometry |
91
+ | `get_era5_climate(lat, lon, year)` | Copernicus CDS (live) or bulk cache | path to EPW file | buildingmodel — weather boundary condition |
87
92
 
88
93
  Reference datasets are pulled from a public Google Cloud Storage bucket and
89
94
  cached locally with generation-based freshness checks. French geospatial data
90
- uses CRS **EPSG:2154 (Lambert-93)**.
95
+ uses CRS **EPSG:2154 (Lambert-93)**. The documentation page *Datasets*
96
+ describes each dataset in detail — provenance, schema, units — and how
97
+ `buildingmodel` / `building_eload` consume it.
91
98
 
92
99
  ## Bulk prefetch for large-scale simulations
93
100
 
@@ -40,18 +40,25 @@ buildingdata configure --bucket my-bucket --cache-dir ~/.cache/buildingdata
40
40
 
41
41
  ## What it provides
42
42
 
43
- | Function | Source | Returns |
44
- | --- | --- | --- |
45
- | `get_census()` | INSEE census (GCS) | polars `DataFrame` |
46
- | `get_districts()` | IRIS geometries (GCS) | geopandas `GeoDataFrame` |
47
- | `get_diagnosis()` | ADEME energy performance diagnoses (GCS) | polars `DataFrame` |
48
- | `get_gas_network()` | GRDF gas network routes (GCS) | geopandas `GeoDataFrame` |
49
- | `get_bdtopo(iris_code)` | IGN Géoplateforme WFS (live) | geopandas `GeoDataFrame` |
50
- | `get_era5_climate(lat, lon, year)` | Copernicus CDS (live) | path to EPW file |
43
+ | Function | Source | Returns | Consumed by |
44
+ | --- | --- | --- | --- |
45
+ | `get_census()` | INSEE census (GCS) | polars `DataFrame` | buildingmodel — dwelling inference |
46
+ | `get_districts()` | IRIS geometries (GCS) | geopandas `GeoDataFrame` | buildingmodel — geocoding; building_eload — district lookup |
47
+ | `get_diagnosis()` | ADEME energy performance diagnoses (GCS) | polars `DataFrame` | buildingmodel — envelope & system inference |
48
+ | `get_gas_network()` | GRDF gas network routes (GCS) | geopandas `GeoDataFrame` | buildingmodel — gas-connection inference |
49
+ | `get_ore()` | Agence ORE annual consumption per IRIS (GCS) | polars `DataFrame` | building_eload — calibration target |
50
+ | `get_enedis_national()` / `get_enedis_regional()` | Enedis conso-inf36 measured consumption (GCS) | polars `DataFrame` | building_eload — validation reference |
51
+ | `get_elmas(table)` | ELMAS non-residential load curves (GCS) | polars `DataFrame` | building_eload — non-residential model |
52
+ | `get_occupant_diaries()` | synthetic occupant activity calendar (GCS) | polars `DataFrame` | building_eload — occupancy models |
53
+ | `get_elecdom()` | Enedis panel Elecdom end-use load curves (GCS) | polars `DataFrame` | provenance only — source of building_eload constants |
54
+ | `get_bdtopo(iris_code)` | IGN Géoplateforme WFS (live) or bulk cache | geopandas `GeoDataFrame` | building_eload → buildingmodel — input geometry |
55
+ | `get_era5_climate(lat, lon, year)` | Copernicus CDS (live) or bulk cache | path to EPW file | buildingmodel — weather boundary condition |
51
56
 
52
57
  Reference datasets are pulled from a public Google Cloud Storage bucket and
53
58
  cached locally with generation-based freshness checks. French geospatial data
54
- uses CRS **EPSG:2154 (Lambert-93)**.
59
+ uses CRS **EPSG:2154 (Lambert-93)**. The documentation page *Datasets*
60
+ describes each dataset in detail — provenance, schema, units — and how
61
+ `buildingmodel` / `building_eload` consume it.
55
62
 
56
63
  ## Bulk prefetch for large-scale simulations
57
64
 
@@ -4,6 +4,7 @@ from .reference import (
4
4
  get_census,
5
5
  get_diagnosis,
6
6
  get_districts,
7
+ get_elecdom,
7
8
  get_elmas,
8
9
  get_enedis_national,
9
10
  get_enedis_regional,
@@ -27,16 +27,31 @@ def get_client():
27
27
  def get_blob(name):
28
28
  """Return the Blob object for a given name, or None if not found.
29
29
 
30
+ Returning None on a missing blob is the contract ``ensure_blob_cached``
31
+ relies on to raise a named
32
+ :class:`~buildingdata.exceptions.RemoteNotAvailableError` -- which is in
33
+ turn what every ``get_X`` accessor documents under ``Raises``. ``reload()``
34
+ is what populates the blob's metadata (``generation``, size) used by the
35
+ cache freshness check, but it raises ``NotFound`` on a missing object
36
+ *before* ``exists()`` is ever reached, so it is caught here. Without this,
37
+ an accessor whose blob has not been staged yet surfaced an opaque
38
+ ``google.api_core.exceptions.NotFound`` instead of the documented error.
39
+
30
40
  Args:
31
41
  name (str): blob name inside the configured bucket.
32
42
 
33
43
  Returns:
34
44
  google.cloud.storage.Blob or None.
35
45
  """
46
+ from google.api_core.exceptions import NotFound
47
+
36
48
  client = get_client()
37
49
  bucket = client.bucket(get_bucket())
38
50
  blob = bucket.blob(name)
39
- blob.reload()
51
+ try:
52
+ blob.reload()
53
+ except NotFound:
54
+ return None
40
55
  return blob if blob.exists() else None
41
56
 
42
57
 
@@ -2,6 +2,7 @@
2
2
  from .census import get_census
3
3
  from .diagnosis import get_diagnosis
4
4
  from .districts import get_districts
5
+ from .elecdom import get_elecdom
5
6
  from .elmas import get_elmas
6
7
  from .enedis import get_enedis_national, get_enedis_regional
7
8
  from .gas_network import get_gas_network
@@ -0,0 +1,108 @@
1
+ # -*- coding: utf-8 -*-
2
+ import polars as pl
3
+
4
+ from ..gcs import download_blob, ensure_blob_cached, get_blob
5
+ from ..validation import require_columns
6
+
7
+
8
+ _BLOB_NAME = "energy_performance_diagnosis_latest.parquet"
9
+
10
+
11
+ # Heating/DHW energies excluded from inference (no meaningful DPE data for coal)
12
+ _EXCLUDED_ENERGIES = ["Charbon"]
13
+
14
+ # Columns this accessor operates on directly (coal filter + dtype casts). If any
15
+ # is absent the raw polars error is opaque; validating up front names the DPE
16
+ # schema drift explicitly. ``heating_system`` + ``region`` are the join keys
17
+ # ``buildingmodel``'s energy-system inference keys on; ``backup_heating_energy``
18
+ # / ``dhw_energy`` carry the fuel labels the coal filter reads.
19
+ _REQUIRED_COLUMNS = (
20
+ "heating_system",
21
+ "region",
22
+ "backup_heating_energy",
23
+ "dhw_energy",
24
+ )
25
+
26
+
27
+ def get_diagnosis(refresh=False):
28
+ """Return the cleaned DPE energy performance diagnosis DataFrame.
29
+
30
+ Downloads energy_performance_diagnosis_latest.parquet from GCS on first
31
+ call. Applies the filtering and type casts that previously lived in
32
+ buildingmodel/io/diagnosis.py so that buildingmodel receives a clean frame.
33
+
34
+ One row is one post-reform DPE record (issued on or after 1 July 2022), not
35
+ one dwelling of the stock: records are matched to buildings by
36
+ ``buildingmodel``, so every column is intensive — a ratio, a U-value, a rate
37
+ or a per-m² quantity — and carries over regardless of the building's size.
38
+
39
+ Args:
40
+ refresh (bool): force re-download even if the cache is warm.
41
+ Defaults to False.
42
+
43
+ Returns:
44
+ polars.DataFrame: DPE records. Beyond the matching keys
45
+ (``construction_year_class``, ``residential_type``,
46
+ ``heating_system``, and the ``district``/``city``/``city_group``/
47
+ ``department``/``region`` geography) the columns group as:
48
+
49
+ * envelope — ``wall_u_value``, ``roof_u_value``, ``floor_u_value``,
50
+ ``wall_window_u_value``, ``wall_window_share``,
51
+ ``envelope_u_value`` (whole-envelope Ubat),
52
+ ``thermal_bridge_linear_loss``, ``thermal_bridge_loss_share``,
53
+ ``air_change_rate``, ``air_permeability``, ``storey_height``,
54
+ ``inertia_class``, ``{wall,roof,floor}_insulation_type``;
55
+ * systems — ``main_heating_energy``, ``backup_heating_energy``,
56
+ ``dhw_energy``, ``heating_mode``/``dhw_mode`` (individual /
57
+ collective / mixed), the ``*_efficiency`` and ``*_scop`` pairs,
58
+ ``intermittency_factor``, ``backup_heating_share``,
59
+ ``dhw_storage_volume``;
60
+ * observed performance — ``energy_class``, ``ghg_class`` and the
61
+ ``annual_*_per_area`` intensities, for calibrating simulated
62
+ output against the diagnosis itself.
63
+
64
+ ``main_heating_system_efficiency`` is a combustion efficiency (≤ 1)
65
+ and is **null for heat pumps**, which have no such value; their
66
+ seasonal performance is in ``main_heating_system_scop`` instead
67
+ (null for every other generator). The same split applies to the
68
+ ``backup_heating_*`` pair. Treating a SCOP as an efficiency
69
+ understates heat-pump performance roughly threefold, hence the two
70
+ columns.
71
+
72
+ Raises:
73
+ RemoteNotAvailableError: if the blob is not found in the GCS bucket.
74
+ SchemaValidationError: if the fetched frame is missing one of the
75
+ columns this accessor operates on (see ``_REQUIRED_COLUMNS``).
76
+ """
77
+ dest = ensure_blob_cached(
78
+ _BLOB_NAME, refresh=refresh, get_blob_fn=get_blob, download_blob_fn=download_blob
79
+ )
80
+
81
+
82
+ df = pl.read_parquet(dest)
83
+
84
+
85
+ require_columns(df, _REQUIRED_COLUMNS, "DPE/EPC diagnosis")
86
+
87
+ # Remove records with coal heating/DHW — no useful inference data.
88
+ # ``fill_null(False)`` is load-bearing: ``backup_heating_energy`` is null for
89
+ # the ~76% of records with no secondary generator, ``is_in`` returns null for
90
+ # those, and ``filter`` drops null rows. Without it this filter keeps only
91
+ # dwellings that happen to own a backup system — a heavily biased subsample —
92
+ # instead of dropping the few hundred coal records it is meant to remove.
93
+ df = df.filter(
94
+ ~pl.col("backup_heating_energy").is_in(_EXCLUDED_ENERGIES).fill_null(False)
95
+ & ~pl.col("dhw_energy").is_in(_EXCLUDED_ENERGIES).fill_null(False)
96
+ )
97
+
98
+ df = df.with_columns([
99
+ pl.col("heating_system").cast(pl.Categorical),
100
+ pl.col("region").cast(pl.Int64),
101
+ ])
102
+
103
+ if "living_area" in df.columns:
104
+ df = df.drop(["living_area"])
105
+ if "living_area_class" in df.columns:
106
+ df = df.drop(["living_area_class"])
107
+
108
+ return df
@@ -0,0 +1,153 @@
1
+ # -*- coding: utf-8 -*-
2
+ import polars as pl
3
+
4
+ from ..exceptions import SchemaValidationError
5
+ from ..gcs import download_blob, ensure_blob_cached, get_blob
6
+ from ..validation import require_columns, require_dtype, require_value_range
7
+
8
+ _BLOB_NAME = "elecdom_daily_load_curves_latest.parquet"
9
+
10
+ _DATASET = "Enedis panel Elecdom daily load curves"
11
+
12
+ # The tidy (long) shape the staged blob must have: one row per
13
+ # (period, end use, hour-of-day). See ``doc/pending_uploads.md`` section 5 for
14
+ # the exact transformation from the source CSV.
15
+ _REQUIRED_COLUMNS = (
16
+ "period_label",
17
+ "season",
18
+ "day_type",
19
+ "end_use",
20
+ "is_specific_electricity",
21
+ "hour",
22
+ "mean_power_w",
23
+ )
24
+
25
+ _STRING_COLUMNS = ("period_label", "season", "day_type", "end_use")
26
+
27
+ # 5 season blocks x 3 day types x 10 end uses x 24 hours.
28
+ _EXPECTED_ROWS = 5 * 3 * 10 * 24
29
+
30
+ _SEASONS = ("december_february", "march_may", "june_august", "september_november", "all_seasons")
31
+ _DAY_TYPES = ("weekday", "weekend", "all_days")
32
+
33
+ # Plausible band for an hourly panel-mean per-dwelling end-use load, in watts.
34
+ # The published curves peak around 410 W; 20 kW leaves ample headroom while
35
+ # still catching a W <-> kWh/Wh-per-day style rescale.
36
+ _MAX_PLAUSIBLE_W = 20_000.0
37
+ # A W -> kW rescale would leave every value below 1 W and would *not* trip the
38
+ # upper bound, so the lower end of the scale is asserted separately.
39
+ _MIN_PLAUSIBLE_PEAK_W = 1.0
40
+
41
+
42
+ def _validate(df):
43
+ """Assert the Elecdom load-curve blob matches its documented contract."""
44
+ require_columns(df, _REQUIRED_COLUMNS, _DATASET)
45
+ require_dtype(df, _STRING_COLUMNS, pl.String, _DATASET)
46
+ require_dtype(df, ("is_specific_electricity",), pl.Boolean, _DATASET)
47
+ require_dtype(df, ("hour",), pl.Int64, _DATASET)
48
+ require_dtype(df, ("mean_power_w",), pl.Float64, _DATASET)
49
+
50
+ if df.height != _EXPECTED_ROWS:
51
+ raise SchemaValidationError(
52
+ f"{_DATASET}: got {df.height} rows, expected exactly {_EXPECTED_ROWS} "
53
+ f"({len(_SEASONS)} season blocks x {len(_DAY_TYPES)} day types x 10 end "
54
+ f"uses x 24 hours). A short frame means the source CSV was partially "
55
+ f"read or a period/end-use block was dropped during staging."
56
+ )
57
+
58
+ hours = sorted(df["hour"].unique().to_list())
59
+ if hours != list(range(24)):
60
+ raise SchemaValidationError(
61
+ f"{_DATASET}: 'hour' must cover 0..23 exactly, got {hours}. The source "
62
+ f"publishes 24 left-closed hourly bins '[h,h+1['."
63
+ )
64
+ for column, expected in (("season", _SEASONS), ("day_type", _DAY_TYPES)):
65
+ observed = sorted(df[column].unique().to_list())
66
+ if observed != sorted(expected):
67
+ raise SchemaValidationError(
68
+ f"{_DATASET}: {column!r} must be exactly {sorted(expected)}, got "
69
+ f"{observed}. The 'Période' parsing at staging time drifted."
70
+ )
71
+
72
+ require_value_range(df, "mean_power_w", 0.0, _MAX_PLAUSIBLE_W, _DATASET)
73
+ peak = df["mean_power_w"].max()
74
+ if peak is None or peak < _MIN_PLAUSIBLE_PEAK_W:
75
+ raise SchemaValidationError(
76
+ f"{_DATASET}: 'mean_power_w' peaks at {peak}, below "
77
+ f"{_MIN_PLAUSIBLE_PEAK_W} W. The published curves peak in the hundreds "
78
+ f"of watts; this looks like a watt-to-kilowatt rescale, which the upper "
79
+ f"bound cannot catch."
80
+ )
81
+ return df
82
+
83
+
84
+ def get_elecdom(refresh=False):
85
+ """Return the Enedis "panel Elecdom" daily end-use load curves (polars).
86
+
87
+ **Provenance-only accessor.** Nothing in ``building_eload`` calls this at
88
+ runtime. The Elecdom study is the published source behind two hand-derived
89
+ literal constant sets in ``building_eload``
90
+ (``models/specific.py::ACTIVITY_ELECTRICITY_WEIGHT_TABLE`` and
91
+ ``models/cooking.py::COOKING_CONFIG``, paper Tab. 1 ref. [37]); those stay
92
+ the runtime source of truth. This accessor exists so the raw table those
93
+ constants were transcribed from is versioned, fetchable and schema-checked,
94
+ making an offline re-derivation reproducible. Its only intended consumer is
95
+ such a re-derivation script.
96
+
97
+ Provenance: Enedis "panel Elecdom" appliance / end-use load study, open
98
+ dataset *"Courbes de charge journalières horaires par usage"*
99
+ (`elecdom-courbes-de-charge-journalieres-horaires`), cited as paper Tab. 1
100
+ ref. [37]. The published table is a panel-mean per-dwelling daily load
101
+ curve, broken down by season block, day type and end use.
102
+
103
+ Shape: tidy/long, one row per (period, end use, hour) -- 5 season blocks
104
+ x 3 day types x 10 end uses x 24 hours = 3600 rows.
105
+
106
+ Columns:
107
+
108
+ * ``period_label`` (``str``) -- the verbatim source ``Période`` label, e.g.
109
+ ``"Mois de Décembre - Février (Jours ouvrées)"``. Kept so a row can be
110
+ traced back to the raw CSV without reversing the parsing.
111
+ * ``season`` (``str``) -- one of ``"december_february"``, ``"march_may"``,
112
+ ``"june_august"``, ``"september_november"``, ``"all_seasons"``.
113
+ * ``day_type`` (``str``) -- ``"weekday"`` (source *Jours ouvrées*),
114
+ ``"weekend"`` (*Week-end*), ``"all_days"`` (source's empty parentheses).
115
+ * ``end_use`` (``str``) -- the verbatim French ``Poste`` label, one of
116
+ ``Chauffage direct`` (direct electric heating), ``Chauffe eau`` (DHW),
117
+ ``Cuisson`` (cooking), ``Ventilation``, ``Froid`` (refrigeration),
118
+ ``Audiovisuel``, ``Informatique/Bureautique``, ``Lavage séchage``
119
+ (washing/drying), ``Eclairage`` (lighting), ``Autres``. Left untranslated
120
+ on purpose: this is a provenance artifact, and any English mapping would
121
+ be an editorial choice the re-derivation script should own.
122
+ * ``is_specific_electricity`` (``bool``) -- True for the seven end uses the
123
+ source also republishes under its ``"Electricité spécifique - "`` period
124
+ prefix (everything except direct heating, DHW and cooking). The prefixed
125
+ block is a byte-identical duplicate of the base rows, so it is collapsed
126
+ into this flag rather than duplicated (verified at staging; see
127
+ ``doc/pending_uploads.md``).
128
+ * ``hour`` (``Int64``) -- 0..23, the left edge of the source's left-closed
129
+ hourly bin ``[h, h+1[``.
130
+ * ``mean_power_w`` (``Float64``) -- **watts**: the panel-mean load over that
131
+ hour bin, equivalently Wh consumed during the hour. Published values run
132
+ from a few W to ~410 W.
133
+
134
+ Args:
135
+ refresh (bool): force re-download even if the cache is warm.
136
+ Defaults to False.
137
+
138
+ Returns:
139
+ polars.DataFrame: the 3600-row tidy Elecdom load-curve table.
140
+
141
+ Raises:
142
+ RemoteNotAvailableError: if the blob is not found in the GCS bucket.
143
+ The blob is **not staged yet** -- see ``doc/pending_uploads.md``
144
+ section 5 for the maintainer handoff -- so this is the expected
145
+ outcome until it is uploaded.
146
+ SchemaValidationError: if the fetched frame is missing a column, carries
147
+ a wrong dtype, is not the full 3600-row grid, has an unexpected
148
+ season/day-type/hour vocabulary, or has implausible watt values.
149
+ """
150
+ dest = ensure_blob_cached(
151
+ _BLOB_NAME, refresh=refresh, get_blob_fn=get_blob, download_blob_fn=download_blob
152
+ )
153
+ return _validate(pl.read_parquet(dest))
@@ -0,0 +1,257 @@
1
+ # -*- coding: utf-8 -*-
2
+ import re
3
+
4
+ import polars as pl
5
+
6
+ from ..exceptions import SchemaValidationError
7
+ from ..gcs import download_blob, ensure_blob_cached, get_blob
8
+ from ..validation import require_dtype
9
+
10
+ _BLOB_NAME = "occupant_diaries_latest.parquet"
11
+
12
+ _DATASET = "occupant activity diaries"
13
+
14
+ # The timeline column. The staged blob was written from pandas, which round-trips
15
+ # an unnamed index through parquet as ``__index_level_0__``; that is the name the
16
+ # blob in the bucket actually carries today, and ``building_eload``'s
17
+ # ``load_activity_diaries`` renames it to ``datetime`` on read. Accept either, so
18
+ # a future re-staging that names the index properly keeps validating, but reject
19
+ # anything else -- that rename is load-bearing for the consumer.
20
+ _INDEX_COLUMN_NAMES = ("datetime", "__index_level_0__")
21
+
22
+ # Every non-timeline column must be an occupant column. ``DwellingProcessor``
23
+ # builds its occupant pool as ``[c for c in columns if c != "datetime"]``, so a
24
+ # stray extra column (a leftover id, a weekday flag) would silently be drawn as
25
+ # if it were an occupant's diary. Pinning the name shape makes that fail loud.
26
+ _OCCUPANT_COLUMN_RE = re.compile(r"^occupant_\d+$")
27
+
28
+ # The activity-code vocabulary. This mirrors ``building_eload``'s
29
+ # ``models/occupancy.py::DEFAULT_OCCUPANCY_PER_ACTIVITY`` (flattened) -- the only
30
+ # place the vocabulary is enumerated. It is transcribed rather than imported
31
+ # because the dependency runs the other way (``building_eload`` depends on
32
+ # ``buildingdata``), so keep the two in sync when the taxonomy changes.
33
+ #
34
+ # Validating it matters because both consumers of an unknown code fail
35
+ # *silently*: ``set_occupancy_labels`` maps anything unrecognised to
36
+ # ``"is_absent"`` (the occupant vanishes from the load), and ``specific.py``'s
37
+ # left join against ``ACTIVITY_ELECTRICITY_WEIGHT_TABLE`` yields a null weight.
38
+ # The check is a subset test, not equality: a smaller diary book that happens
39
+ # not to use every code is legitimate.
40
+ _ACTIVITY_CODES = frozenset(
41
+ {
42
+ # at home
43
+ "Hygiene",
44
+ "Other_at_home",
45
+ "Meal_at_home",
46
+ "Home_working",
47
+ "Cooking",
48
+ "Dish_washing",
49
+ "Housework",
50
+ "Clothes_washing",
51
+ "Ironing",
52
+ "Home_DIY",
53
+ "Gardening",
54
+ "Television",
55
+ "HiFi",
56
+ "Video_game",
57
+ "Computer",
58
+ "Electronic_Device",
59
+ # asleep
60
+ "Sleep",
61
+ # away
62
+ "Other_outside",
63
+ "Meal_outside",
64
+ "At_work",
65
+ "At_school",
66
+ "Shopping",
67
+ "Work_journey",
68
+ "Other_journey",
69
+ }
70
+ )
71
+
72
+ _SECONDS_PER_DAY = 24 * 60 * 60
73
+ _FULL_YEAR_DAYS = (365, 366)
74
+
75
+
76
+ def _index_column(df):
77
+ """Return the name of the diary's single Datetime timeline column."""
78
+ datetime_cols = [name for name, dtype in df.schema.items() if dtype == pl.Datetime]
79
+ if not datetime_cols:
80
+ raise SchemaValidationError(
81
+ f"{_DATASET}: no Datetime column found; expected exactly one timeline "
82
+ f"column named one of {list(_INDEX_COLUMN_NAMES)}. Got dtypes "
83
+ f"{ {name: str(t) for name, t in list(df.schema.items())[:5]} } (first 5 "
84
+ f"of {df.width}). Without a timeline the diaries cannot be aligned "
85
+ f"onto a simulation year."
86
+ )
87
+ if len(datetime_cols) > 1:
88
+ raise SchemaValidationError(
89
+ f"{_DATASET}: expected exactly one Datetime timeline column, found "
90
+ f"{datetime_cols}. Consumers treat every non-timeline column as an "
91
+ f"occupant diary, so a second temporal column would be drawn as one."
92
+ )
93
+ name = datetime_cols[0]
94
+ if name not in _INDEX_COLUMN_NAMES:
95
+ raise SchemaValidationError(
96
+ f"{_DATASET}: the timeline column is named {name!r}, expected one of "
97
+ f"{list(_INDEX_COLUMN_NAMES)}. ``building_eload`` renames "
98
+ f"``__index_level_0__`` to ``datetime`` on read and would raise on "
99
+ f"any other name."
100
+ )
101
+ return name
102
+
103
+
104
+ def _require_complete_timeline(df, index_col):
105
+ """Assert the timeline is a gap-free regular grid covering a whole year.
106
+
107
+ The ELMAS lesson applies here too: a diary book that is one step short, has
108
+ a duplicated timestamp, or covers ten months instead of twelve does not
109
+ raise anywhere downstream -- ``align_activity_diaries`` rolls and slices it
110
+ happily and produces a quietly wrong calendar.
111
+ """
112
+ series = df[index_col]
113
+ if series.null_count():
114
+ raise SchemaValidationError(
115
+ f"{_DATASET}: the {index_col!r} timeline has "
116
+ f"{series.null_count()} null timestamp(s)."
117
+ )
118
+ if not series.is_sorted():
119
+ raise SchemaValidationError(
120
+ f"{_DATASET}: the {index_col!r} timeline is not sorted ascending; "
121
+ f"consumers roll the frame positionally and assume calendar order."
122
+ )
123
+
124
+ steps = series.diff().drop_nulls().unique().to_list()
125
+ if len(steps) != 1:
126
+ raise SchemaValidationError(
127
+ f"{_DATASET}: the {index_col!r} timeline is not a regular grid; found "
128
+ f"{len(steps)} distinct steps {sorted(steps)[:5]} over {df.height} rows. "
129
+ f"A gap or a duplicated timestamp silently misaligns the calendar roll."
130
+ )
131
+ step = steps[0]
132
+ step_seconds = step.total_seconds()
133
+ if step_seconds <= 0 or _SECONDS_PER_DAY % step_seconds:
134
+ raise SchemaValidationError(
135
+ f"{_DATASET}: timeline step {step} does not divide a calendar day "
136
+ f"evenly; the diaries cannot be indexed by day-of-week."
137
+ )
138
+
139
+ steps_per_day = int(_SECONDS_PER_DAY // step_seconds)
140
+ days, remainder = divmod(df.height, steps_per_day)
141
+ if remainder or days not in _FULL_YEAR_DAYS:
142
+ raise SchemaValidationError(
143
+ f"{_DATASET}: {df.height} rows at a {step} step is {df.height / steps_per_day:.4f} "
144
+ f"days, not a whole year ({' or '.join(str(d) for d in _FULL_YEAR_DAYS)} days, "
145
+ f"i.e. {' or '.join(str(d * steps_per_day) for d in _FULL_YEAR_DAYS)} rows). "
146
+ f"A truncated diary book is the exact failure mode this check exists for: "
147
+ f"it raises nothing downstream, it just yields a wrong year."
148
+ )
149
+
150
+
151
+ def _require_known_activity_codes(df, occupant_cols):
152
+ """Assert every observed activity code is in the documented vocabulary."""
153
+ observed = set()
154
+ for column in occupant_cols:
155
+ observed.update(df[column].unique().drop_nulls().to_list())
156
+ unknown = sorted(observed - _ACTIVITY_CODES)
157
+ if unknown:
158
+ raise SchemaValidationError(
159
+ f"{_DATASET}: {len(unknown)} unknown activity code(s) {unknown[:10]}"
160
+ f"{'...' if len(unknown) > 10 else ''}. The documented vocabulary is "
161
+ f"{sorted(_ACTIVITY_CODES)}. Unknown codes do not raise downstream -- "
162
+ f"``set_occupancy_labels`` maps them to 'is_absent' and the specific-"
163
+ f"electricity weight join yields null -- so they are rejected here. If "
164
+ f"the taxonomy genuinely grew, extend both ``_ACTIVITY_CODES`` here and "
165
+ f"``building_eload``'s DEFAULT_OCCUPANCY_PER_ACTIVITY / "
166
+ f"ACTIVITY_ELECTRICITY_WEIGHT_TABLE."
167
+ )
168
+
169
+
170
+ def _validate(df):
171
+ """Run the full occupant-diary contract check on a loaded frame."""
172
+ index_col = _index_column(df)
173
+
174
+ occupant_cols = [c for c in df.columns if c != index_col]
175
+ if not occupant_cols:
176
+ raise SchemaValidationError(
177
+ f"{_DATASET}: no occupant column found next to the {index_col!r} "
178
+ f"timeline; the diary book is empty."
179
+ )
180
+ malformed = [c for c in occupant_cols if not _OCCUPANT_COLUMN_RE.match(c)]
181
+ if malformed:
182
+ raise SchemaValidationError(
183
+ f"{_DATASET}: {len(malformed)} non-timeline column(s) do not match "
184
+ f"'occupant_<n>': {malformed[:10]}{'...' if len(malformed) > 10 else ''}. "
185
+ f"``DwellingProcessor`` draws every non-timeline column as an occupant "
186
+ f"profile, so a stray column would be simulated as a person."
187
+ )
188
+
189
+ require_dtype(df, occupant_cols, pl.String, _DATASET)
190
+
191
+ null_counts = df.select(pl.col(occupant_cols).null_count()).row(0)
192
+ nulls = [c for c, n in zip(occupant_cols, null_counts) if n]
193
+ if nulls:
194
+ raise SchemaValidationError(
195
+ f"{_DATASET}: {len(nulls)} occupant column(s) contain null activity "
196
+ f"codes: {nulls[:10]}{'...' if len(nulls) > 10 else ''}. A null code is "
197
+ f"treated as 'is_absent' downstream without warning."
198
+ )
199
+
200
+ _require_complete_timeline(df, index_col)
201
+ _require_known_activity_codes(df, occupant_cols)
202
+ return df
203
+
204
+
205
+ def get_occupant_diaries(refresh=False):
206
+ """Return the occupant activity-diary calendar DataFrame (polars).
207
+
208
+ The occupant-diary "book" is a synthetic, time-indexed activity calendar:
209
+ one column per simulated occupant holding an activity-code string per
210
+ timestep, consumed by ``building_eload``'s ``DynamicSimulation`` as the base
211
+ pool of activity profiles that dwellings draw from. Downloaded from
212
+ ``occupant_diaries_latest.parquet`` on GCS on first call.
213
+
214
+ The canonical blob is the 1000-occupant book that ``DynamicParameters``
215
+ uses by default (``occupant_book_1000_pers_*.parquet``). Larger books
216
+ (10k/100k occupants) are separate artifacts staged by the maintainer and
217
+ passed explicitly via ``DynamicParameters.activity_file``; they are not
218
+ served here.
219
+
220
+ Contract enforced on every call (see ``Raises``):
221
+
222
+ * exactly one naive ``Datetime`` timeline column, named ``datetime`` or
223
+ ``__index_level_0__`` (the pandas-written index name the staged blob
224
+ currently carries; ``building_eload`` renames it to ``datetime``);
225
+ * at least one occupant column, and *every* non-timeline column named
226
+ ``occupant_<n>`` -- consumers draw all of them as occupant profiles;
227
+ * every occupant column ``String``-typed and null-free;
228
+ * the timeline a gap-free, ascending, regular grid whose step divides a
229
+ day, spanning exactly one whole calendar year (365 or 366 days);
230
+ * every observed activity code in the documented vocabulary (the flattened
231
+ ``building_eload`` ``DEFAULT_OCCUPANCY_PER_ACTIVITY`` taxonomy, 24 codes).
232
+ The check is a subset test -- a book that does not use every code passes.
233
+
234
+ Vintage note: the blob staged today is a **10-minute** book over leap year
235
+ 2024 (52 704 rows x 1000 occupants). The step is validated generically
236
+ rather than pinned to 10 minutes, but note that ``building_eload``'s
237
+ ``align_activity_diaries`` hardcodes ``intervals_per_day = 144``: restaging
238
+ at another resolution requires a matching change there.
239
+
240
+ Args:
241
+ refresh (bool): force re-download even if the cache is warm.
242
+ Defaults to False.
243
+
244
+ Returns:
245
+ polars.DataFrame: one ``occupant_<i>`` column per occupant (``str``
246
+ activity codes) plus a naive ``Datetime`` timeline column, over one
247
+ full representative calendar year.
248
+
249
+ Raises:
250
+ RemoteNotAvailableError: if the blob is not found in the GCS bucket.
251
+ SchemaValidationError: if the fetched frame violates any clause of the
252
+ contract above, naming the offending column(s) or code(s).
253
+ """
254
+ dest = ensure_blob_cached(
255
+ _BLOB_NAME, refresh=refresh, get_blob_fn=get_blob, download_blob_fn=download_blob
256
+ )
257
+ return _validate(pl.read_parquet(dest))