buildingdata 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {buildingdata-0.3.0 → buildingdata-0.4.0}/PKG-INFO +1 -1
- buildingdata-0.4.0/buildingdata/reference/diagnosis.py +108 -0
- buildingdata-0.4.0/buildingdata/tests/test_pipeline_diagnosis.py +283 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/test_reference.py +26 -9
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata.egg-info/PKG-INFO +1 -1
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata.egg-info/SOURCES.txt +1 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/pyproject.toml +1 -1
- buildingdata-0.3.0/buildingdata/reference/diagnosis.py +0 -74
- {buildingdata-0.3.0 → buildingdata-0.4.0}/LICENSE +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/README.md +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/__init__.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/_cli.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/bulk.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/cache.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/config.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/exceptions.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/gcs.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/__init__.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/census.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/districts.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/elmas.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/enedis.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/gas_network.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/occupant_diaries.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/reference/ore.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/simulation/__init__.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/simulation/_epw.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/simulation/bdtopo.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/simulation/bdtopo_bulk.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/simulation/era5.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/simulation/era5_bulk.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/__init__.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/conftest.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/test_bulk.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/test_cache.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/test_config.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/test_public_api.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/tests/test_simulation.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata/validation.py +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata.egg-info/dependency_links.txt +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata.egg-info/entry_points.txt +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata.egg-info/requires.txt +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/buildingdata.egg-info/top_level.txt +0 -0
- {buildingdata-0.3.0 → buildingdata-0.4.0}/setup.cfg +0 -0
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
import polars as pl
|
|
3
|
+
|
|
4
|
+
from ..gcs import download_blob, ensure_blob_cached, get_blob
|
|
5
|
+
from ..validation import require_columns
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
_BLOB_NAME = "energy_performance_diagnosis_latest.parquet"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# Heating/DHW energies excluded from inference (no meaningful DPE data for coal)
|
|
12
|
+
_EXCLUDED_ENERGIES = ["Charbon"]
|
|
13
|
+
|
|
14
|
+
# Columns this accessor operates on directly (coal filter + dtype casts). If any
|
|
15
|
+
# is absent the raw polars error is opaque; validating up front names the DPE
|
|
16
|
+
# schema drift explicitly. ``heating_system`` + ``region`` are the join keys
|
|
17
|
+
# ``buildingmodel``'s energy-system inference keys on; ``backup_heating_energy``
|
|
18
|
+
# / ``dhw_energy`` carry the fuel labels the coal filter reads.
|
|
19
|
+
_REQUIRED_COLUMNS = (
|
|
20
|
+
"heating_system",
|
|
21
|
+
"region",
|
|
22
|
+
"backup_heating_energy",
|
|
23
|
+
"dhw_energy",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def get_diagnosis(refresh=False):
|
|
28
|
+
"""Return the cleaned DPE energy performance diagnosis DataFrame.
|
|
29
|
+
|
|
30
|
+
Downloads energy_performance_diagnosis_latest.parquet from GCS on first
|
|
31
|
+
call. Applies the filtering and type casts that previously lived in
|
|
32
|
+
buildingmodel/io/diagnosis.py so that buildingmodel receives a clean frame.
|
|
33
|
+
|
|
34
|
+
One row is one post-reform DPE record (issued on or after 1 July 2022), not
|
|
35
|
+
one dwelling of the stock: records are matched to buildings by
|
|
36
|
+
``buildingmodel``, so every column is intensive — a ratio, a U-value, a rate
|
|
37
|
+
or a per-m² quantity — and carries over regardless of the building's size.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
refresh (bool): force re-download even if the cache is warm.
|
|
41
|
+
Defaults to False.
|
|
42
|
+
|
|
43
|
+
Returns:
|
|
44
|
+
polars.DataFrame: DPE records. Beyond the matching keys
|
|
45
|
+
(``construction_year_class``, ``residential_type``,
|
|
46
|
+
``heating_system``, and the ``district``/``city``/``city_group``/
|
|
47
|
+
``department``/``region`` geography) the columns group as:
|
|
48
|
+
|
|
49
|
+
* envelope — ``wall_u_value``, ``roof_u_value``, ``floor_u_value``,
|
|
50
|
+
``wall_window_u_value``, ``wall_window_share``,
|
|
51
|
+
``envelope_u_value`` (whole-envelope Ubat),
|
|
52
|
+
``thermal_bridge_linear_loss``, ``thermal_bridge_loss_share``,
|
|
53
|
+
``air_change_rate``, ``air_permeability``, ``storey_height``,
|
|
54
|
+
``inertia_class``, ``{wall,roof,floor}_insulation_type``;
|
|
55
|
+
* systems — ``main_heating_energy``, ``backup_heating_energy``,
|
|
56
|
+
``dhw_energy``, ``heating_mode``/``dhw_mode`` (individual /
|
|
57
|
+
collective / mixed), the ``*_efficiency`` and ``*_scop`` pairs,
|
|
58
|
+
``intermittency_factor``, ``backup_heating_share``,
|
|
59
|
+
``dhw_storage_volume``;
|
|
60
|
+
* observed performance — ``energy_class``, ``ghg_class`` and the
|
|
61
|
+
``annual_*_per_area`` intensities, for calibrating simulated
|
|
62
|
+
output against the diagnosis itself.
|
|
63
|
+
|
|
64
|
+
``main_heating_system_efficiency`` is a combustion efficiency (≤ 1)
|
|
65
|
+
and is **null for heat pumps**, which have no such value; their
|
|
66
|
+
seasonal performance is in ``main_heating_system_scop`` instead
|
|
67
|
+
(null for every other generator). The same split applies to the
|
|
68
|
+
``backup_heating_*`` pair. Treating a SCOP as an efficiency
|
|
69
|
+
understates heat-pump performance roughly threefold, hence the two
|
|
70
|
+
columns.
|
|
71
|
+
|
|
72
|
+
Raises:
|
|
73
|
+
RemoteNotAvailableError: if the blob is not found in the GCS bucket.
|
|
74
|
+
SchemaValidationError: if the fetched frame is missing one of the
|
|
75
|
+
columns this accessor operates on (see ``_REQUIRED_COLUMNS``).
|
|
76
|
+
"""
|
|
77
|
+
dest = ensure_blob_cached(
|
|
78
|
+
_BLOB_NAME, refresh=refresh, get_blob_fn=get_blob, download_blob_fn=download_blob
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
df = pl.read_parquet(dest)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
require_columns(df, _REQUIRED_COLUMNS, "DPE/EPC diagnosis")
|
|
86
|
+
|
|
87
|
+
# Remove records with coal heating/DHW — no useful inference data.
|
|
88
|
+
# ``fill_null(False)`` is load-bearing: ``backup_heating_energy`` is null for
|
|
89
|
+
# the ~76% of records with no secondary generator, ``is_in`` returns null for
|
|
90
|
+
# those, and ``filter`` drops null rows. Without it this filter keeps only
|
|
91
|
+
# dwellings that happen to own a backup system — a heavily biased subsample —
|
|
92
|
+
# instead of dropping the few hundred coal records it is meant to remove.
|
|
93
|
+
df = df.filter(
|
|
94
|
+
~pl.col("backup_heating_energy").is_in(_EXCLUDED_ENERGIES).fill_null(False)
|
|
95
|
+
& ~pl.col("dhw_energy").is_in(_EXCLUDED_ENERGIES).fill_null(False)
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
df = df.with_columns([
|
|
99
|
+
pl.col("heating_system").cast(pl.Categorical),
|
|
100
|
+
pl.col("region").cast(pl.Int64),
|
|
101
|
+
])
|
|
102
|
+
|
|
103
|
+
if "living_area" in df.columns:
|
|
104
|
+
df = df.drop(["living_area"])
|
|
105
|
+
if "living_area_class" in df.columns:
|
|
106
|
+
df = df.drop(["living_area_class"])
|
|
107
|
+
|
|
108
|
+
return df
|
|
@@ -0,0 +1,283 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""Contracts of pipeline/rules/scripts/process_diagnosis.py.
|
|
3
|
+
|
|
4
|
+
The script is not part of the installed distribution, so it is loaded from the
|
|
5
|
+
repo by path and skipped when absent (e.g. running the suite from a wheel).
|
|
6
|
+
|
|
7
|
+
What is pinned here is what silently drifts: the value vocabularies that must
|
|
8
|
+
stay identical to process_census.py for buildingmodel's join to hit, the
|
|
9
|
+
ordering-sensitive generator classification, and the efficiency/SCOP split.
|
|
10
|
+
"""
|
|
11
|
+
import ast
|
|
12
|
+
import importlib.util
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
import polars as pl
|
|
16
|
+
import pytest
|
|
17
|
+
|
|
18
|
+
_SCRIPT = (
|
|
19
|
+
Path(__file__).resolve().parents[2]
|
|
20
|
+
/ "pipeline"
|
|
21
|
+
/ "rules"
|
|
22
|
+
/ "scripts"
|
|
23
|
+
/ "process_diagnosis.py"
|
|
24
|
+
)
|
|
25
|
+
_CENSUS_SCRIPT = _SCRIPT.with_name("process_census.py")
|
|
26
|
+
|
|
27
|
+
pytestmark = pytest.mark.skipif(
|
|
28
|
+
not _SCRIPT.exists(), reason="pipeline/ is not shipped in the distribution"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _load(path):
|
|
33
|
+
spec = importlib.util.spec_from_file_location(path.stem, path)
|
|
34
|
+
module = importlib.util.module_from_spec(spec)
|
|
35
|
+
spec.loader.exec_module(module)
|
|
36
|
+
return module
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _read_literal(path, name):
|
|
40
|
+
"""Read a module-level literal without importing the module.
|
|
41
|
+
|
|
42
|
+
process_census.py reads ``snakemake`` and imports buildingmodel at module
|
|
43
|
+
level, so it cannot be imported here.
|
|
44
|
+
"""
|
|
45
|
+
tree = ast.parse(path.read_text(encoding="utf-8"))
|
|
46
|
+
for node in tree.body:
|
|
47
|
+
targets = getattr(node, "targets", [])
|
|
48
|
+
if any(getattr(t, "id", None) == name for t in targets):
|
|
49
|
+
return ast.literal_eval(node.value)
|
|
50
|
+
raise AssertionError(f"{name} not found in {path}")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@pytest.fixture(scope="module")
|
|
54
|
+
def script():
|
|
55
|
+
return _load(_SCRIPT)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# One coherent upstream record. Tests override only the fields they exercise, so
|
|
59
|
+
# a new column in COLUMN_SELECTION shows up as a KeyError here rather than as a
|
|
60
|
+
# silently-null column in the published dataset.
|
|
61
|
+
UPSTREAM_DEFAULTS = {
|
|
62
|
+
# identity / matching keys
|
|
63
|
+
"type_energie_chauffage_principal": "Gaz naturel",
|
|
64
|
+
"periode_construction_recensement": "1946 à 1970",
|
|
65
|
+
"type_logement": "appartement",
|
|
66
|
+
"surface_habitable_logement": 100.0,
|
|
67
|
+
"SURF": "de 80 à 100 m²",
|
|
68
|
+
"zone_climatique": "H1a",
|
|
69
|
+
"classe_altitude": "inférieur à 400m",
|
|
70
|
+
# envelope
|
|
71
|
+
"classe_inertie": "Moyenne",
|
|
72
|
+
"u_mur": 1.2,
|
|
73
|
+
"u_ph": 0.5,
|
|
74
|
+
"u_pb": 1.7,
|
|
75
|
+
"uw_baie": 2.6,
|
|
76
|
+
"ubat_calcule": 0.95,
|
|
77
|
+
"k_pont_thermique": 0.5,
|
|
78
|
+
"q4pa_conv": 1.9,
|
|
79
|
+
"hsp": 2.5,
|
|
80
|
+
"qvarep_conv": 1.35,
|
|
81
|
+
"surface_paroi_opaque_mur": 85.0,
|
|
82
|
+
"surface_baie_vertical_Nord": 3.0,
|
|
83
|
+
"surface_baie_vertical_Sud": 4.0,
|
|
84
|
+
"surface_baie_vertical_Est": 2.0,
|
|
85
|
+
"surface_baie_vertical_Ouest": 2.0,
|
|
86
|
+
"deperdition_pont_thermique": 20.0,
|
|
87
|
+
"deperdition_enveloppe": 200.0,
|
|
88
|
+
"type_isolation_majoritaire_mur": "ITI",
|
|
89
|
+
"type_isolation_majoritaire_ph": "Non isolé",
|
|
90
|
+
"type_isolation_majoritaire_pb": "inconnu",
|
|
91
|
+
# heating system 1
|
|
92
|
+
"type_generateur_ch_systeme_1": "Chaudière gaz à condensation après 2015",
|
|
93
|
+
"mode_chauffage": "individuel",
|
|
94
|
+
"surface_chauffee_systeme_1": 100.0,
|
|
95
|
+
"rendement_generation_systeme_1": 0.9,
|
|
96
|
+
"scop_systeme_1": None,
|
|
97
|
+
"rendement_emission_systeme_1": 0.95,
|
|
98
|
+
"rendement_distribution_systeme_1": 0.95,
|
|
99
|
+
"rendement_regulation_systeme_1": 0.99,
|
|
100
|
+
"i0_systeme_1": 0.9,
|
|
101
|
+
"conso_ch_systeme_1": 8000.0,
|
|
102
|
+
# heating system 2 (a wood stove backing up the boiler)
|
|
103
|
+
"type_energie_systeme_2": "Bois – Bûches",
|
|
104
|
+
"rendement_generation_systeme_2": 0.5,
|
|
105
|
+
"scop_systeme_2": None,
|
|
106
|
+
"rendement_emission_systeme_2": 0.95,
|
|
107
|
+
"rendement_distribution_systeme_2": 1.0,
|
|
108
|
+
"rendement_regulation_systeme_2": 0.99,
|
|
109
|
+
"conso_ch_systeme_2": 2000.0,
|
|
110
|
+
# DHW
|
|
111
|
+
"type_energie_ecs": "Électricité",
|
|
112
|
+
"mode_ecs": "individuel",
|
|
113
|
+
"rendement_generation_ecs": 0.8,
|
|
114
|
+
"rendement_generation_stockage_ecs": 0.0,
|
|
115
|
+
"rendement_stockage_ecs": 0.0,
|
|
116
|
+
"volume_stockage_ecs": 150.0,
|
|
117
|
+
# observed performance
|
|
118
|
+
"classe_bilan_dpe": "D",
|
|
119
|
+
"classe_emission_ges": "C",
|
|
120
|
+
"besoin_ch": 10000.0,
|
|
121
|
+
"besoin_ecs": 1700.0,
|
|
122
|
+
"conso_5_usages_m2": 145.0,
|
|
123
|
+
"ep_conso_5_usages_m2": 230.0,
|
|
124
|
+
"emission_ges_5_usages_m2": 18.0,
|
|
125
|
+
# geography
|
|
126
|
+
"code_iris": "751124705",
|
|
127
|
+
"code_insee": "75112",
|
|
128
|
+
"DEP": "75",
|
|
129
|
+
"EPCI": "200054781",
|
|
130
|
+
"REG": "11",
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _upstream(*overrides):
|
|
135
|
+
"""Build an upstream frame, one row per override dict."""
|
|
136
|
+
rows = [{**UPSTREAM_DEFAULTS, **o} for o in (overrides or [{}])]
|
|
137
|
+
return pl.DataFrame(
|
|
138
|
+
{key: [row[key] for row in rows] for key in UPSTREAM_DEFAULTS},
|
|
139
|
+
schema_overrides={
|
|
140
|
+
key: pl.Float64
|
|
141
|
+
for key, value in UPSTREAM_DEFAULTS.items()
|
|
142
|
+
if isinstance(value, float)
|
|
143
|
+
},
|
|
144
|
+
).lazy()
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _run(script, *overrides):
|
|
148
|
+
return script.transform(_upstream(*overrides)).collect()
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def test_a_fuel_generator_gets_a_chained_efficiency_and_no_scop(script):
|
|
152
|
+
out = _run(script)
|
|
153
|
+
|
|
154
|
+
assert out["main_heating_system_efficiency"][0] == pytest.approx(
|
|
155
|
+
0.9 * 0.95 * 0.95 * 0.99
|
|
156
|
+
)
|
|
157
|
+
assert out["main_heating_system_scop"][0] is None
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_a_heat_pump_gets_a_scop_and_no_efficiency(script):
|
|
161
|
+
# ADEME leaves rendement_generation null and fills scop for every heat pump.
|
|
162
|
+
# Publishing the SCOP as an efficiency would understate it ~3x, so the two
|
|
163
|
+
# live in separate columns.
|
|
164
|
+
out = _run(
|
|
165
|
+
script,
|
|
166
|
+
{
|
|
167
|
+
"type_generateur_ch_systeme_1": "PAC air/eau installée après 2017",
|
|
168
|
+
"rendement_generation_systeme_1": None,
|
|
169
|
+
"scop_systeme_1": 3.0,
|
|
170
|
+
},
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
assert out["heating_system"][0] == "electric heat pump"
|
|
174
|
+
assert out["main_heating_system_efficiency"][0] is None
|
|
175
|
+
assert out["main_heating_system_scop"][0] == pytest.approx(3.0 * 0.95 * 0.95 * 0.99)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def test_a_multi_building_gas_plant_is_classified_as_a_heat_network(script):
|
|
179
|
+
# "Chaudière(s) gaz multi bâtiment modélisée comme un réseau de chaleur"
|
|
180
|
+
# matches both a gas and a heat-network pattern. ADEME reports the fuel as
|
|
181
|
+
# "Réseau de Chauffage urbain" for these, so the heat-network reading is the
|
|
182
|
+
# one that keeps main_heating_energy and heating_system coherent — which is
|
|
183
|
+
# only true because "réseau de chaleur" is tested first.
|
|
184
|
+
out = _run(
|
|
185
|
+
script,
|
|
186
|
+
{
|
|
187
|
+
"type_generateur_ch_systeme_1": (
|
|
188
|
+
"Chaudière(s) gaz multi bâtiment modélisée comme un réseau de chaleur"
|
|
189
|
+
),
|
|
190
|
+
"type_energie_chauffage_principal": "Réseau de Chauffage urbain",
|
|
191
|
+
},
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
assert out["heating_system"][0] == "urban heat network"
|
|
195
|
+
assert out["main_heating_energy"][0] == "district_network"
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def test_an_unclassifiable_generator_is_dropped(script):
|
|
199
|
+
# Coal boilers and cooking ranges used as main heating: buildingmodel has no
|
|
200
|
+
# category for them, so they must not reach the join.
|
|
201
|
+
assert _run(script, {"type_generateur_ch_systeme_1": "Cuisinière installé avant 1990"}).height == 0
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def test_the_generator_vocabulary_matches_the_census_one(script):
|
|
205
|
+
# buildingmodel joins buildings (census heating_system) to diagnosis records
|
|
206
|
+
# on this column: a label present on one side only silently loses those rows.
|
|
207
|
+
census_labels = set(
|
|
208
|
+
_read_literal(_CENSUS_SCRIPT, "HEATING_SYSTEM_LABELS").values()
|
|
209
|
+
) - {"unknown"}
|
|
210
|
+
|
|
211
|
+
assert set(script.HEATING_SYSTEM_PATTERNS.values()) == census_labels
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def test_coal_survives_the_fuel_relabelling(script):
|
|
215
|
+
# get_diagnosis() drops coal records by matching the literal "Charbon", so
|
|
216
|
+
# this script must not translate it away.
|
|
217
|
+
out = _run(script, {"type_energie_ecs": "Charbon"})
|
|
218
|
+
|
|
219
|
+
assert out["dhw_energy"][0] == "Charbon"
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def test_houses_get_the_system_mode_upstream_leaves_null(script):
|
|
223
|
+
# methode_application_dpe_log only spells out a mode for multi-dwelling
|
|
224
|
+
# buildings, so every house arrives null.
|
|
225
|
+
individual, networked = _run(
|
|
226
|
+
script,
|
|
227
|
+
{"type_logement": "maison", "mode_chauffage": None, "mode_ecs": None},
|
|
228
|
+
{
|
|
229
|
+
"type_logement": "maison",
|
|
230
|
+
"mode_chauffage": None,
|
|
231
|
+
"mode_ecs": None,
|
|
232
|
+
"type_energie_chauffage_principal": "Réseau de Chauffage urbain",
|
|
233
|
+
"type_generateur_ch_systeme_1": "Réseau de chaleur isolé",
|
|
234
|
+
},
|
|
235
|
+
)["heating_mode"]
|
|
236
|
+
|
|
237
|
+
assert individual == "individual"
|
|
238
|
+
assert networked == "collective"
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def test_an_apartment_keeps_its_upstream_system_mode(script):
|
|
242
|
+
out = _run(script, {"mode_chauffage": "collectif", "mode_ecs": "mixte"})
|
|
243
|
+
|
|
244
|
+
assert out["heating_mode"][0] == "collective"
|
|
245
|
+
assert out["dhw_mode"][0] == "mixed"
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def test_an_out_of_range_energy_intensity_is_nulled_not_clipped(script):
|
|
249
|
+
# These are calibration targets: clipping would pull a fit toward the bound,
|
|
250
|
+
# nulling just drops the record from the comparison.
|
|
251
|
+
out = _run(script, {"besoin_ch": 10_000_000.0})
|
|
252
|
+
|
|
253
|
+
assert out["annual_heating_need_per_area"][0] is None
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def test_an_out_of_range_physical_attribute_is_clipped_not_nulled(script):
|
|
257
|
+
out = _run(script, {"hsp": 3018.0, "k_pont_thermique": 275.0})
|
|
258
|
+
|
|
259
|
+
assert out["storey_height"][0] == script.STOREY_HEIGHT_BOUNDS[1]
|
|
260
|
+
assert out["thermal_bridge_linear_loss"][0] == (
|
|
261
|
+
script.THERMAL_BRIDGE_LINEAR_LOSS_BOUNDS[1]
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def test_a_record_with_an_implausible_glazing_share_is_dropped(script):
|
|
266
|
+
# A wall reported as almost entirely glass is a DPE geometry-coding error.
|
|
267
|
+
assert _run(script, {"surface_paroi_opaque_mur": 1.0}).height == 0
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def test_every_published_column_is_populated(script):
|
|
271
|
+
# Guards COLUMN_SELECTION against a source rename: a missing source column
|
|
272
|
+
# would otherwise surface as an all-null column in the published dataset.
|
|
273
|
+
out = _run(script)
|
|
274
|
+
|
|
275
|
+
assert set(out.columns) == set(script.COLUMN_SELECTION)
|
|
276
|
+
# The SCOP columns are null by construction for a non-heat-pump record.
|
|
277
|
+
scop_columns = {"main_heating_system_scop", "backup_heating_system_scop"}
|
|
278
|
+
unexpected_nulls = [
|
|
279
|
+
column
|
|
280
|
+
for column in out.columns
|
|
281
|
+
if column not in scop_columns and out[column].null_count()
|
|
282
|
+
]
|
|
283
|
+
assert unexpected_nulls == []
|
|
@@ -130,15 +130,17 @@ def test_get_census_refresh_redownloads(monkeypatch):
|
|
|
130
130
|
# --- get_diagnosis ----------------------------------------------------------
|
|
131
131
|
|
|
132
132
|
def _diagnosis_df():
|
|
133
|
+
# Row 3 has no secondary generator, which is the majority case in the real
|
|
134
|
+
# dataset (~76% of DPE records): backup_heating_energy is null there.
|
|
133
135
|
return pl.DataFrame(
|
|
134
136
|
{
|
|
135
|
-
"heating_system": ["gas_boiler", "heat_pump", "gas_boiler"],
|
|
136
|
-
"region": [11, 84, 11],
|
|
137
|
-
"backup_heating_energy": ["Gaz", "Charbon", "Electricite"],
|
|
138
|
-
"dhw_energy": ["Gaz", "Gaz", "Charbon"],
|
|
139
|
-
"living_area": [80.0, 95.0, 60.0],
|
|
140
|
-
"living_area_class": ["A", "B", "A"],
|
|
141
|
-
"u_wall": [0.3, 0.4, 0.5],
|
|
137
|
+
"heating_system": ["gas_boiler", "heat_pump", "gas_boiler", "gas_boiler"],
|
|
138
|
+
"region": [11, 84, 11, 11],
|
|
139
|
+
"backup_heating_energy": ["Gaz", "Charbon", "Electricite", None],
|
|
140
|
+
"dhw_energy": ["Gaz", "Gaz", "Charbon", "Gaz"],
|
|
141
|
+
"living_area": [80.0, 95.0, 60.0, 70.0],
|
|
142
|
+
"living_area_class": ["A", "B", "A", "B"],
|
|
143
|
+
"u_wall": [0.3, 0.4, 0.5, 0.6],
|
|
142
144
|
}
|
|
143
145
|
)
|
|
144
146
|
|
|
@@ -149,8 +151,8 @@ def test_get_diagnosis_filters_coal_and_casts(monkeypatch):
|
|
|
149
151
|
|
|
150
152
|
out = diagnosis_mod.get_diagnosis()
|
|
151
153
|
|
|
152
|
-
# Two of
|
|
153
|
-
assert out.height ==
|
|
154
|
+
# Two of four rows reference coal (backup or dhw) and must be dropped.
|
|
155
|
+
assert out.height == 2
|
|
154
156
|
assert out["heating_system"].dtype == pl.Categorical
|
|
155
157
|
assert out["region"].dtype == pl.Int64
|
|
156
158
|
# living_area / living_area_class are dropped.
|
|
@@ -160,6 +162,21 @@ def test_get_diagnosis_filters_coal_and_casts(monkeypatch):
|
|
|
160
162
|
assert "u_wall" in out.columns
|
|
161
163
|
|
|
162
164
|
|
|
165
|
+
def test_get_diagnosis_keeps_records_without_a_backup_system(monkeypatch):
|
|
166
|
+
# Regression: `~col.is_in([...])` is null when the column is null, and
|
|
167
|
+
# `filter` drops null rows — so the coal filter used to discard every record
|
|
168
|
+
# with no secondary generator (80% of the dataset, and a biased subsample:
|
|
169
|
+
# only dwellings owning a backup wood stove or similar survived).
|
|
170
|
+
df = _diagnosis_df().with_columns(
|
|
171
|
+
pl.lit(None, dtype=pl.String).alias("backup_heating_energy"),
|
|
172
|
+
pl.lit("Gaz").alias("dhw_energy"),
|
|
173
|
+
)
|
|
174
|
+
blob = FakeBlob(diagnosis_mod._BLOB_NAME, generation=2)
|
|
175
|
+
_stub_gcs(monkeypatch, diagnosis_mod, blob, lambda dest: df.write_parquet(dest))
|
|
176
|
+
|
|
177
|
+
assert diagnosis_mod.get_diagnosis().height == df.height
|
|
178
|
+
|
|
179
|
+
|
|
163
180
|
def test_get_diagnosis_raises_when_blob_absent(monkeypatch):
|
|
164
181
|
monkeypatch.setattr(diagnosis_mod, "get_blob", lambda name: None)
|
|
165
182
|
with pytest.raises(RemoteNotAvailableError):
|
|
@@ -35,6 +35,7 @@ buildingdata/tests/conftest.py
|
|
|
35
35
|
buildingdata/tests/test_bulk.py
|
|
36
36
|
buildingdata/tests/test_cache.py
|
|
37
37
|
buildingdata/tests/test_config.py
|
|
38
|
+
buildingdata/tests/test_pipeline_diagnosis.py
|
|
38
39
|
buildingdata/tests/test_public_api.py
|
|
39
40
|
buildingdata/tests/test_reference.py
|
|
40
41
|
buildingdata/tests/test_simulation.py
|
|
@@ -1,74 +0,0 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
import polars as pl
|
|
3
|
-
|
|
4
|
-
from ..gcs import download_blob, ensure_blob_cached, get_blob
|
|
5
|
-
from ..validation import require_columns
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
_BLOB_NAME = "energy_performance_diagnosis_latest.parquet"
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
# Heating/DHW energies excluded from inference (no meaningful DPE data for coal)
|
|
12
|
-
_EXCLUDED_ENERGIES = ["Charbon"]
|
|
13
|
-
|
|
14
|
-
# Columns this accessor operates on directly (coal filter + dtype casts). If any
|
|
15
|
-
# is absent the raw polars error is opaque; validating up front names the DPE
|
|
16
|
-
# schema drift explicitly. ``heating_system`` + ``region`` are the join keys
|
|
17
|
-
# ``buildingmodel``'s energy-system inference keys on; ``backup_heating_energy``
|
|
18
|
-
# / ``dhw_energy`` carry the fuel labels the coal filter reads.
|
|
19
|
-
_REQUIRED_COLUMNS = (
|
|
20
|
-
"heating_system",
|
|
21
|
-
"region",
|
|
22
|
-
"backup_heating_energy",
|
|
23
|
-
"dhw_energy",
|
|
24
|
-
)
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
def get_diagnosis(refresh=False):
|
|
28
|
-
"""Return the cleaned DPE energy performance diagnosis DataFrame.
|
|
29
|
-
|
|
30
|
-
Downloads energy_performance_diagnosis_latest.parquet from GCS on first
|
|
31
|
-
call. Applies the filtering and type casts that previously lived in
|
|
32
|
-
buildingmodel/io/diagnosis.py so that buildingmodel receives a clean frame.
|
|
33
|
-
|
|
34
|
-
Args:
|
|
35
|
-
refresh (bool): force re-download even if the cache is warm.
|
|
36
|
-
Defaults to False.
|
|
37
|
-
|
|
38
|
-
Returns:
|
|
39
|
-
polars.DataFrame: DPE records with columns heating_system (Categorical),
|
|
40
|
-
region (Int64), backup_heating_energy, dhw_energy, and all U-value
|
|
41
|
-
and efficiency columns used by inference/building_attributes.py.
|
|
42
|
-
|
|
43
|
-
Raises:
|
|
44
|
-
RemoteNotAvailableError: if the blob is not found in the GCS bucket.
|
|
45
|
-
SchemaValidationError: if the fetched frame is missing one of the
|
|
46
|
-
columns this accessor operates on (see ``_REQUIRED_COLUMNS``).
|
|
47
|
-
"""
|
|
48
|
-
dest = ensure_blob_cached(
|
|
49
|
-
_BLOB_NAME, refresh=refresh, get_blob_fn=get_blob, download_blob_fn=download_blob
|
|
50
|
-
)
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
df = pl.read_parquet(dest)
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
require_columns(df, _REQUIRED_COLUMNS, "DPE/EPC diagnosis")
|
|
57
|
-
|
|
58
|
-
# Remove records with coal heating/DHW — no useful inference data
|
|
59
|
-
df = df.filter(
|
|
60
|
-
~pl.col("backup_heating_energy").is_in(_EXCLUDED_ENERGIES)
|
|
61
|
-
& ~pl.col("dhw_energy").is_in(_EXCLUDED_ENERGIES)
|
|
62
|
-
)
|
|
63
|
-
|
|
64
|
-
df = df.with_columns([
|
|
65
|
-
pl.col("heating_system").cast(pl.Categorical),
|
|
66
|
-
pl.col("region").cast(pl.Int64),
|
|
67
|
-
])
|
|
68
|
-
|
|
69
|
-
if "living_area" in df.columns:
|
|
70
|
-
df = df.drop(["living_area"])
|
|
71
|
-
if "living_area_class" in df.columns:
|
|
72
|
-
df = df.drop(["living_area_class"])
|
|
73
|
-
|
|
74
|
-
return df
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|