shrecc 0.1.2.dev2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. {shrecc-0.1.2.dev2/shrecc.egg-info → shrecc-0.2.0}/PKG-INFO +7 -1
  2. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/README.md +5 -0
  3. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/pyproject.toml +1 -0
  4. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/__init__.py +1 -1
  5. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/tyndp/tyndp_connections.csv +5 -5
  6. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/tyndp/tyndp_countries.csv +3 -3
  7. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/database.py +4 -3
  8. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/energy_charts.py +6 -2
  9. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/pipeline.py +43 -5
  10. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/premise_mapping.py +18 -2
  11. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/result_store.py +7 -1
  12. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/solver.py +33 -7
  13. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/tyndp.py +74 -12
  14. {shrecc-0.1.2.dev2 → shrecc-0.2.0/shrecc.egg-info}/PKG-INFO +7 -1
  15. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc.egg-info/requires.txt +1 -0
  16. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_database.py +3 -3
  17. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_energy_charts_api.py +47 -0
  18. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_pipeline.py +96 -0
  19. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_solver.py +58 -0
  20. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_tyndp_scenario.py +68 -0
  21. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/LICENSE +0 -0
  22. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/MANIFEST.in +0 -0
  23. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/setup.cfg +0 -0
  24. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/_legacy_database.py +0 -0
  25. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/_legacy_treatment.py +0 -0
  26. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/_legacy_tyndp.py +0 -0
  27. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/activity_mapping.py +0 -0
  28. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/analysis.py +0 -0
  29. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/__init__.py +0 -0
  30. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/el_map_all_norm.csv +0 -0
  31. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/el_map_all_norm_w_ned.csv +0 -0
  32. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/generation_units_by_country.csv +0 -0
  33. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/techs_agg.json +0 -0
  34. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/tyndp/__init__.py +0 -0
  35. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/tyndp/remind-eu-topology.json +0 -0
  36. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/data/tyndp/tyndp_activities.csv +0 -0
  37. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/download.py +0 -0
  38. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/lcia.py +0 -0
  39. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/mapping.py +0 -0
  40. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/treatment.py +0 -0
  41. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc/treatment_fiona.py +0 -0
  42. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc.egg-info/SOURCES.txt +0 -0
  43. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc.egg-info/dependency_links.txt +0 -0
  44. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/shrecc.egg-info/top_level.txt +0 -0
  45. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_analysis.py +0 -0
  46. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_architecture.py +0 -0
  47. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_energy_charts.py +0 -0
  48. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_lcia.py +0 -0
  49. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_mapping.py +0 -0
  50. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_premise_mapping.py +0 -0
  51. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_result_store.py +0 -0
  52. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_treatment.py +0 -0
  53. {shrecc-0.1.2.dev2 → shrecc-0.2.0}/tests/test_tyndp_paths_download.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: shrecc
3
- Version: 0.1.2.dev2
3
+ Version: 0.2.0
4
4
  Summary: SHRECC: Smooth Hourly Resolution Electricity Consumption Calculation
5
5
  Author-email: Sabina Bednářová <sabina.bednarova@list.lu>
6
6
  Maintainer-email: Sabina Bednářová <sabina.bednarova@list.lu>
@@ -34,6 +34,7 @@ Requires-Dist: packaging
34
34
  Requires-Dist: tqdm
35
35
  Requires-Dist: xarray
36
36
  Requires-Dist: python-calamine>=0.6.2
37
+ Requires-Dist: openpyxl
37
38
  Provides-Extra: testing
38
39
  Requires-Dist: pytest>=8; extra == "testing"
39
40
  Requires-Dist: pytest-cov; extra == "testing"
@@ -204,3 +205,8 @@ Licensed under the MIT License.
204
205
 
205
206
  * Sabina Bednářová (<sabina.bednarova@list.lu>)
206
207
  * Thomas Gibon (<thomas.gibon@list.lu>)
208
+
209
+ ### Contributors
210
+
211
+ + Isabela PICHARDO VELAZQUEZ <isabela.pichardo@list.lu>
212
+ + Caipeng LIANG <caipeng.liang@list.lu>
@@ -134,3 +134,8 @@ Licensed under the MIT License.
134
134
 
135
135
  * Sabina Bednářová (<sabina.bednarova@list.lu>)
136
136
  * Thomas Gibon (<thomas.gibon@list.lu>)
137
+
138
+ ### Contributors
139
+
140
+ + Isabela PICHARDO VELAZQUEZ <isabela.pichardo@list.lu>
141
+ + Caipeng LIANG <caipeng.liang@list.lu>
@@ -44,6 +44,7 @@ dependencies = [
44
44
  "tqdm",
45
45
  "xarray",
46
46
  "python-calamine>=0.6.2",
47
+ "openpyxl",
47
48
  ]
48
49
  license="MIT"
49
50
  license-files = ["LICENSE"]
@@ -19,4 +19,4 @@ from .energy_charts import data_processing, get_data
19
19
  from .lcia import LCIAResults
20
20
  from .pipeline import NewDatabase
21
21
 
22
- __version__ = "0.1.2.dev2"
22
+ __version__ = "0.2.0"
@@ -19,8 +19,8 @@ BE00-FR00,BE00,FR00,BE,FR,BE-FR
19
19
  BE00-LUB1,BE00,LUB1,BE,LU,BE-LU
20
20
  BE00-LUG1,BE00,LUG1,BE,LU,BE-LU
21
21
  BE00-NL00,BE00,NL00,BE,NL,BE-NL
22
- BE00-BEOF,BE00,BEOF,BE,NW,BE-NW
23
- BE00-NL60,BE00,NL60,BE,NW,BE-NW
22
+ BE00-BEOF,BE00,BEOF,BE,BE,BE-BE
23
+ BE00-NL60,BE00,NL60,BE,NL,BE-NL
24
24
  BE00-UK00,BE00,UK00,BE,UK,BE-UK
25
25
  BG00-GR00,BG00,GR00,BG,GR,BG-GR
26
26
  BG00-MK00,BG00,MK00,BG,MK,BG-MK
@@ -56,7 +56,7 @@ DE00-SE04,DE00,SE04,DE,SE,DE-SE
56
56
  DE00-UK00,DE00,UK00,DE,UK,DE-UK
57
57
  DKW1-NL00,DKW1,NL00,DK,NL,DK-NL
58
58
  DKW1-NOS0,DKW1,NOS0,DK,NO,DK-NO
59
- DKNS-DKW1,DKNS,DKW1,NW,DK,DK-NW
59
+ DKNS-DKW1,DKNS,DKW1,DK,DK,DK-DK
60
60
  DKE1-PL00,DKE1,PL00,DK,PL,DK-PL
61
61
  DKE1-SE04,DKE1,SE04,DK,SE,DK-SE
62
62
  DKW1-SE03,DKW1,SE03,DK,SE,DK-SE
@@ -124,7 +124,7 @@ MK00-RS00,MK00,RS00,MK,RS,MK-RS
124
124
  MK00-XK00,MK00,XK00,MK,XK,MK-XK
125
125
  MT00-TN00,MT00,TN00,MT,TN,MT-TN
126
126
  NL00-NOS0,NL00,NOS0,NL,NO,NL-NO
127
- NL00-NL60,NL00,NL60,NL,NW,NL-NW
127
+ NL00-NL60,NL00,NL60,NL,NL,NL-NL
128
128
  NL00-UK00,NL00,UK00,NL,UK,NL-UK
129
129
  NLLL-UK00,NLLL,UK00,NL,UK,NL-UK
130
130
  XRU00-NON1,XRU00,NON1,RU,NO,NO-RU
@@ -133,7 +133,7 @@ NON1-SE01,NON1,SE01,NO,SE,NO-SE
133
133
  NON1-SE02,NON1,SE02,NO,SE,NO-SE
134
134
  NOS0-SE03,NOS0,SE03,NO,SE,NO-SE
135
135
  NOS0-UK00,NOS0,UK00,NO,UK,NO-UK
136
- BEOF-UK00,BEOF,UK00,NW,UK,NW-UK
136
+ BEOF-UK00,BEOF,UK00,BE,UK,BE-UK
137
137
  PL00-SE04,PL00,SE04,PL,SE,PL-SE
138
138
  PL00-SK00,PL00,SK00,PL,SK,PL-SK
139
139
  PL00-UA00,PL00,UA00,PL,UA,PL-UA
@@ -58,9 +58,9 @@ NLLL,NL,Netherlands
58
58
  NOM1,NO,Norway
59
59
  NON1,NO,Norway
60
60
  NOS0,NO,Norway
61
- BEOF,NW,North Hub
62
- DKNS,NW,North Hub
63
- NL60,NW,North Hub
61
+ BEOF,BE,North Hub
62
+ DKNS,DK,North Hub
63
+ NL60,NL,North Hub
64
64
  PL00,PL,Poland
65
65
  PLE0,PL,Poland
66
66
  PLI0,PL,Poland
@@ -99,7 +99,7 @@ def build_background_activity_index(eidb_name):
99
99
 
100
100
  def filt_cutoff(
101
101
  countries,
102
- times=[],
102
+ times=None,
103
103
  general_range=0,
104
104
  refined_range=0,
105
105
  freq=0,
@@ -111,7 +111,6 @@ def filt_cutoff(
111
111
  Filters data based on selected countries and times (either one-off, a range, or periodical range).
112
112
 
113
113
  Args:
114
- year (int): Selected year of the downloaded data.
115
114
  countries (list of str): Countries selected by the user for their database.
116
115
  E.g. countries=['FR', 'DE'].
117
116
  times (list of str): Selecting one specific time, e.g. times = ['2023-06-16 8:00:00', '2023-06-16 22:00:00'].
@@ -131,6 +130,8 @@ def filt_cutoff(
131
130
  Returns:
132
131
  pd.DataFrame: The filtered dataframe.
133
132
  """
133
+ times = [] if times is None else list(times)
134
+
134
135
  if path_to_data:
135
136
  print(f"Using mapping root: {path_to_data}")
136
137
  path_to_data = Path(path_to_data)
@@ -657,7 +658,7 @@ def create_database(
657
658
  inventory_resolution=inventory_resolution,
658
659
  )
659
660
  elec_db = setup_database(project_name, db_name)
660
- elec_db.write(activities)
661
+ elec_db.write(activities, signal=True)
661
662
 
662
663
 
663
664
  def _format_inventory_period(time, inventory_resolution):
@@ -436,6 +436,7 @@ def get_data(
436
436
  request_timeout=API_REQUEST_TIMEOUT,
437
437
  request_interval=2,
438
438
  required_countries=None,
439
+ force_refresh=False,
439
440
  ):
440
441
  """
441
442
  Main function for downloading data.
@@ -449,6 +450,9 @@ def get_data(
449
450
  request_interval (float): Delay between successful endpoint requests.
450
451
  required_countries: Optional country codes which must be available in
451
452
  the completed download.
453
+ force_refresh: Ignore an existing yearly snapshot and download every
454
+ country again. This is used when a current-year cache predates the
455
+ requested time selection.
452
456
 
453
457
  Returns:
454
458
  pd.DataFrame: A dataframe containing both production and trade data for all countries in the selected year.
@@ -465,7 +469,7 @@ def get_data(
465
469
  str(country).upper() for country in (required_countries or [])
466
470
  }
467
471
 
468
- cache_exists = filename.exists()
472
+ cache_exists = filename.exists() and not force_refresh
469
473
  if cache_exists:
470
474
  data = load_from_pickle(filename)
471
475
  if not isinstance(data, dict):
@@ -477,7 +481,7 @@ def get_data(
477
481
  print("Legacy API data loaded successfully.")
478
482
  return cleaning_data(data, files("shrecc.data"))
479
483
 
480
- status = _load_energy_charts_download_status(status_filename)
484
+ status = {} if force_refresh else _load_energy_charts_download_status(status_filename)
481
485
  for country in data:
482
486
  status[str(country).upper()] = "success"
483
487
 
@@ -45,6 +45,7 @@ from shrecc.premise_mapping import (
45
45
  premise_activity_mix_to_database_table,
46
46
  )
47
47
  from shrecc.result_store import (
48
+ CacheTimeSelectionError,
48
49
  MANIFEST_FILENAME,
49
50
  consumption_result_cache_path,
50
51
  get_package_user_data_dir,
@@ -603,11 +604,48 @@ class NewDatabase:
603
604
  include_consumption_mix_volume=False,
604
605
  )
605
606
  time_range, times = self._selection_for_year(year)
606
- results = load_consumption_result_cache(
607
- cache_dir,
608
- general_range=time_range,
609
- times=times,
610
- )
607
+ try:
608
+ results = load_consumption_result_cache(
609
+ cache_dir,
610
+ general_range=time_range,
611
+ times=times,
612
+ )
613
+ except CacheTimeSelectionError as exc:
614
+ if not self.download:
615
+ raise CacheTimeSelectionError(
616
+ f"The Energy Charts cache for {year} does not cover the "
617
+ "requested time selection and download=False prevents it "
618
+ "from being refreshed"
619
+ ) from exc
620
+ if self.verbose:
621
+ print(
622
+ f"Energy Charts cache for {year} does not cover the "
623
+ "requested times; refreshing source data."
624
+ )
625
+ data = get_energy_charts_data(
626
+ year,
627
+ path_to_data=data_root,
628
+ required_countries=self.countries,
629
+ force_refresh=True,
630
+ )
631
+ shutil.rmtree(cache_dir)
632
+ cache_dir = process_energy_charts_data(
633
+ data,
634
+ year,
635
+ path_to_data=data_root,
636
+ include_consumption_mix_volume=False,
637
+ )
638
+ try:
639
+ results = load_consumption_result_cache(
640
+ cache_dir,
641
+ general_range=time_range,
642
+ times=times,
643
+ )
644
+ except CacheTimeSelectionError as refreshed_exc:
645
+ raise CacheTimeSelectionError(
646
+ f"Energy Charts data for {year} still does not cover the "
647
+ "requested time selection after refreshing the cache"
648
+ ) from refreshed_exc
611
649
  missing_result_countries = set(self.countries).difference(
612
650
  results["consumption_mix"]["consumer_country"].to_index()
613
651
  )
@@ -855,11 +855,20 @@ def map_consumption_mix_technologies_xr(
855
855
  )
856
856
 
857
857
  if check:
858
+ # The mapping must conserve shares: compare against what came in, not a
859
+ # hardcoded 1. Whether the mix itself sums to 1 is the solver's invariant
860
+ # (solve_consumption_system checks it with a boundary-country-aware
861
+ # tolerance); asserting it again here would re-fail on inputs the
862
+ # solver has already legitimately accepted.
858
863
  total_dims = [
859
864
  dim for dim in ("source_country", "premise_activity") if dim in mapped.dims
860
865
  ]
866
+ input_dims = [
867
+ dim for dim in ("source_country", "technology") if dim in consumption_mix_xr.dims
868
+ ]
861
869
  totals = mapped.sum(total_dims)
862
- np.testing.assert_allclose(totals.to_numpy(), 1, atol=1e-8)
870
+ expected = consumption_mix_xr.sum(input_dims).transpose(*totals.dims)
871
+ np.testing.assert_allclose(totals.to_numpy(), expected.to_numpy(), atol=1e-6)
863
872
 
864
873
  return mapped
865
874
 
@@ -1005,13 +1014,20 @@ def map_consumption_mix_regions_xr(
1005
1014
  mapped.attrs["iam_model"] = iam_model
1006
1015
 
1007
1016
  if check:
1017
+ # Same conservation-of-input check as map_consumption_mix_technologies_xr.
1008
1018
  total_dims = [
1009
1019
  dim
1010
1020
  for dim in (region_dim, "premise_activity", "technology")
1011
1021
  if dim in mapped.dims
1012
1022
  ]
1023
+ input_dims = [
1024
+ dim
1025
+ for dim in (source_dim, "premise_activity", "technology")
1026
+ if dim in consumption_mix_xr.dims
1027
+ ]
1013
1028
  totals = mapped.sum(total_dims)
1014
- np.testing.assert_allclose(totals.to_numpy(), 1, atol=1e-8)
1029
+ expected = consumption_mix_xr.sum(input_dims).transpose(*totals.dims)
1030
+ np.testing.assert_allclose(totals.to_numpy(), expected.to_numpy(), atol=1e-8)
1015
1031
 
1016
1032
  return mapped
1017
1033
 
@@ -17,6 +17,10 @@ MANIFEST_FILENAME = "manifest.json"
17
17
  CANONICAL_CACHE_DIRNAME = f"consumption_results_v{CACHE_VERSION}"
18
18
 
19
19
 
20
+ class CacheTimeSelectionError(ValueError):
21
+ """Raised when a canonical cache does not cover the requested times."""
22
+
23
+
20
24
  def save_pickle(obj, filename):
21
25
  """Persist a Python object using the repository's legacy pickle format."""
22
26
  filename = Path(filename)
@@ -173,7 +177,9 @@ def iter_consumption_result_cache(
173
177
  yield results
174
178
 
175
179
  if not selected_any:
176
- raise ValueError("The requested time selection contains no cached results")
180
+ raise CacheTimeSelectionError(
181
+ "The requested time selection contains no cached results"
182
+ )
177
183
 
178
184
 
179
185
  def load_consumption_result_cache(
@@ -78,6 +78,7 @@ def solve_consumption_system(
78
78
 
79
79
  if consumption_volume is None:
80
80
  consumption = node_supply - exports
81
+ external_boundary_countries = np.zeros(len(countries), dtype=bool)
81
82
  else:
82
83
  _validate_input_array(
83
84
  consumption_volume,
@@ -88,6 +89,15 @@ def solve_consumption_system(
88
89
  time=times,
89
90
  consumer_country=country_values,
90
91
  )
92
+ # Trade-only nodes (e.g. external neighbours with no domestic demand
93
+ # data) will be NaN after reindex. Fill them with zero: these nodes
94
+ # have no tracked production or domestic consumption in the dataset,
95
+ # so their consumption is zero by definition. They are recorded so the
96
+ # zero-supply check below can skip them.
97
+ missing_mask = consumption_volume.isnull().any(dim="time")
98
+ external_boundary_countries = missing_mask.values
99
+ if missing_mask.any():
100
+ consumption_volume = consumption_volume.fillna(0.0)
91
101
  if consumption_volume.isnull().any():
92
102
  raise ValueError(
93
103
  "consumption_volume does not cover all solver times and countries"
@@ -133,8 +143,14 @@ def solve_consumption_system(
133
143
 
134
144
  direct_total = direct_production.sum(axis=2) + direct_trade.sum(axis=2)
135
145
  if check:
136
- if zero_supply_mask.any() and zero_consumption == "raise":
137
- positions = np.argwhere(zero_supply_mask)
146
+ # External boundary countries (trade-only nodes zero-filled above) are
147
+ # excluded from the zero-supply check — they have no tracked supply by
148
+ # design and are not a data quality issue.
149
+ unhandled_zero_supply = zero_supply_mask & ~external_boundary_countries[None, :]
150
+ if unresolved_technology is not None:
151
+ unhandled_zero_supply = unhandled_zero_supply & ~unresolved_origin_mask
152
+ if unhandled_zero_supply.any() and zero_consumption == "raise":
153
+ positions = np.argwhere(unhandled_zero_supply)
138
154
  examples = [
139
155
  f"{pd.Timestamp(times.values[t])} / {countries[c]}"
140
156
  for t, c in positions[:10]
@@ -191,15 +207,25 @@ def solve_consumption_system(
191
207
 
192
208
  if check:
193
209
  expected_mix_total = np.ones_like(node_supply)
194
- if zero_consumption == "keep_zero":
195
- has_unresolved_origin = np.zeros_like(zero_supply_mask)
196
- if unresolved_technology is not None:
197
- has_unresolved_origin = unresolved_origin_mask
210
+ # Zero-supply country-hours produce a mix sum of 0 in every mode except
211
+ # "month_hour_average" (which imputes them) and "unresolved_technology"
212
+ # (which adds a phantom supply entry). Build a single mask covering all
213
+ # modes rather than branching per mode.
214
+ has_unresolved_origin = np.zeros_like(zero_supply_mask)
215
+ if unresolved_technology is not None:
216
+ has_unresolved_origin = unresolved_origin_mask
217
+ if zero_consumption != "month_hour_average":
198
218
  expected_mix_total[zero_supply_mask & ~has_unresolved_origin] = 0
219
+ # External boundary countries (e.g. NT's RU/SA) are trade-only nodes
220
+ # with zero-filled demand. Their presence indicates the scenario's
221
+ # demand data may not balance the physical supply/trade totals, so the
222
+ # mix-sum tolerance is relaxed to 0.1 to accommodate that inconsistency
223
+ # while still catching gross solver errors.
224
+ mix_atol = 0.1 if external_boundary_countries.any() else 1e-8
199
225
  np.testing.assert_allclose(
200
226
  consumption_mix.sum(axis=(2, 3)),
201
227
  expected_mix_total,
202
- atol=1e-8,
228
+ atol=mix_atol,
203
229
  )
204
230
 
205
231
  mix_dims = (
@@ -986,17 +986,18 @@ def _extract_tyndp_workbook_from_zip(zip_file, workbook, verbose=False):
986
986
  archive_name = names_by_basename.get(workbook.name)
987
987
 
988
988
  if archive_name is None:
989
- xlsb_files = [
989
+ excel_files = [
990
990
  name
991
991
  for name in archive.namelist()
992
- if Path(name).suffix.lower() == ".xlsb"
992
+ if Path(name).suffix.lower() in (".xlsb", ".xlsx")
993
+ and not Path(name).name.startswith(".")
993
994
  ]
994
- if len(xlsb_files) != 1:
995
+ if len(excel_files) != 1:
995
996
  raise FileNotFoundError(
996
997
  f"Could not find {workbook.name!r} in {zip_file}. "
997
- f"Found XLSB files: {xlsb_files}"
998
+ f"Found Excel files: {excel_files}"
998
999
  )
999
- archive_name = xlsb_files[0]
1000
+ archive_name = excel_files[0]
1000
1001
 
1001
1002
  with archive.open(archive_name) as source, workbook.open("wb") as target:
1002
1003
  target.write(source.read())
@@ -1022,7 +1023,39 @@ def _read_excel_dataframe(filename, **kwargs):
1022
1023
  if "engine" in read_kwargs and read_kwargs["engine"] is None:
1023
1024
  read_kwargs.pop("engine")
1024
1025
 
1025
- return cast(pd.DataFrame, pd.read_excel(filename, **read_kwargs))
1026
+ try:
1027
+ return cast(pd.DataFrame, pd.read_excel(filename, **read_kwargs))
1028
+ except ImportError as exc:
1029
+ # Some engines (e.g. "calamine") are optional. NT workbooks are .xlsx
1030
+ # files that do not require calamine, so retry without a specific
1031
+ # engine and let pandas choose the default openpyxl-based reader.
1032
+ if "engine" not in read_kwargs:
1033
+ raise
1034
+ read_kwargs.pop("engine")
1035
+ return cast(pd.DataFrame, pd.read_excel(filename, **read_kwargs))
1036
+ except ValueError as exc:
1037
+ sheet_name = read_kwargs.get("sheet_name")
1038
+ if not isinstance(sheet_name, str) or "not found" not in str(exc).lower():
1039
+ raise
1040
+ # TYNDP workbooks aren't consistently named across scenarios (e.g. the
1041
+ # "Global Ambition" workbook uses "Hourly Market Data emarket" while
1042
+ # "National Trends" just uses "Hourly Market Data"). Fall back to a
1043
+ # sheet whose name is a prefix/suffix match of the requested one
1044
+ # before giving up.
1045
+ available = pd.ExcelFile(filename, engine=read_kwargs.get("engine")).sheet_names
1046
+ candidates = [
1047
+ name
1048
+ for name in available
1049
+ if sheet_name.startswith(name) or name.startswith(sheet_name)
1050
+ ]
1051
+ if len(candidates) != 1:
1052
+ raise ValueError(
1053
+ f"Worksheet named {sheet_name!r} not found in {filename}, and no "
1054
+ f"unambiguous fallback match among {available}"
1055
+ + (f" (candidates: {candidates})" if candidates else "")
1056
+ ) from exc
1057
+ read_kwargs["sheet_name"] = candidates[0]
1058
+ return cast(pd.DataFrame, pd.read_excel(filename, **read_kwargs))
1026
1059
 
1027
1060
 
1028
1061
  def _log(message, verbose):
@@ -1038,6 +1071,9 @@ def _parse_tyndp_datetime_index(index, model_year):
1038
1071
  second level stores labels such as ``"01Jan00:00"`` without a year. If the
1039
1072
  pickle already contains datetimes, they are returned unchanged.
1040
1073
 
1074
+ NT workbooks use all-caps month abbreviations (``"01SEP00:00"``); these are
1075
+ normalised to title-case before parsing so that ``%b`` matches correctly.
1076
+
1041
1077
  Args:
1042
1078
  index: Original TYNDP index.
1043
1079
  model_year: Year appended to labels that do not already contain one.
@@ -1054,15 +1090,34 @@ def _parse_tyndp_datetime_index(index, model_year):
1054
1090
  if pd.api.types.is_datetime64_any_dtype(raw_index):
1055
1091
  return pd.DatetimeIndex(raw_index)
1056
1092
 
1057
- labels = raw_index.astype(str)
1093
+ labels = raw_index.astype(str).str.strip()
1058
1094
 
1059
1095
  if labels.str.contains(r"\d{4}", regex=True).all():
1060
1096
  return pd.to_datetime(labels)
1061
1097
 
1062
- return pd.to_datetime(
1063
- labels + str(model_year),
1064
- format="%d%b%H:%M%Y",
1065
- )
1098
+ # Map month abbreviations to their numeric value instead of relying on
1099
+ # strptime's locale-dependent %b directive (English locale variants don't
1100
+ # agree on abbreviations -- e.g. "English (Ireland)" uses "Sept" for
1101
+ # September instead of the standard "Sep", which silently breaks %b).
1102
+ _MONTH_NUMBERS = {
1103
+ "JAN": 1, "FEB": 2, "MAR": 3, "APR": 4, "MAY": 5, "JUN": 6,
1104
+ "JUL": 7, "AUG": 8, "SEP": 9, "OCT": 10, "NOV": 11, "DEC": 12,
1105
+ }
1106
+
1107
+ def _normalise(label):
1108
+ upper = label.upper()
1109
+ for month, number in _MONTH_NUMBERS.items():
1110
+ if month in upper:
1111
+ day_str, rest = upper.split(month, 1)
1112
+ return f"{day_str}{number:02d}{rest}"
1113
+ return label
1114
+
1115
+ labels = labels.map(_normalise)
1116
+
1117
+ # Parse without the year (pandas' Cython strptime is strict about %M width
1118
+ # when %Y is appended), then replace the placeholder year with model_year.
1119
+ parsed = pd.to_datetime(labels, format="%d%m%H:%M")
1120
+ return parsed.map(lambda t: t.replace(year=model_year))
1066
1121
 
1067
1122
 
1068
1123
  def load_technology_concordance(filename, sheet_name):
@@ -1314,11 +1369,18 @@ def _aggregate_tyndp_trade(trade, connections):
1314
1369
 
1315
1370
 
1316
1371
  def _clean_tyndp_connection_labels(labels):
1317
- """Remove TYNDP concept suffixes from cross-border connection labels."""
1372
+ """Remove TYNDP concept suffixes from cross-border connection labels.
1373
+
1374
+ Also normalises the from/to separator: some TYNDP scenario workbooks
1375
+ (e.g. "National Trends") export connection labels as ``"AL00->GR00"``,
1376
+ while others (e.g. "Global Ambition") and the ``connections`` mapping
1377
+ both use ``"AL00-GR00"``.
1378
+ """
1318
1379
  return (
1319
1380
  pd.Index(labels)
1320
1381
  .astype(str)
1321
1382
  .str.replace(r"\s*/?(Concept|Real)\s+\d+\s*$", "", regex=True)
1383
+ .str.replace("->", "-", regex=False)
1322
1384
  .str.strip()
1323
1385
  )
1324
1386
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: shrecc
3
- Version: 0.1.2.dev2
3
+ Version: 0.2.0
4
4
  Summary: SHRECC: Smooth Hourly Resolution Electricity Consumption Calculation
5
5
  Author-email: Sabina Bednářová <sabina.bednarova@list.lu>
6
6
  Maintainer-email: Sabina Bednářová <sabina.bednarova@list.lu>
@@ -34,6 +34,7 @@ Requires-Dist: packaging
34
34
  Requires-Dist: tqdm
35
35
  Requires-Dist: xarray
36
36
  Requires-Dist: python-calamine>=0.6.2
37
+ Requires-Dist: openpyxl
37
38
  Provides-Extra: testing
38
39
  Requires-Dist: pytest>=8; extra == "testing"
39
40
  Requires-Dist: pytest-cov; extra == "testing"
@@ -204,3 +205,8 @@ Licensed under the MIT License.
204
205
 
205
206
  * Sabina Bednářová (<sabina.bednarova@list.lu>)
206
207
  * Thomas Gibon (<thomas.gibon@list.lu>)
208
+
209
+ ### Contributors
210
+
211
+ + Isabela PICHARDO VELAZQUEZ <isabela.pichardo@list.lu>
212
+ + Caipeng LIANG <caipeng.liang@list.lu>
@@ -10,6 +10,7 @@ packaging
10
10
  tqdm
11
11
  xarray
12
12
  python-calamine>=0.6.2
13
+ openpyxl
13
14
 
14
15
  [:python_version >= "3.13"]
15
16
  numpy>2
@@ -2007,7 +2007,7 @@ def test_create_database_with_network_true(
2007
2007
  consumption_profile="national_demand",
2008
2008
  inventory_resolution=None,
2009
2009
  )
2010
- mock_db.write.assert_called_once_with(activities)
2010
+ mock_db.write.assert_called_once_with(activities, signal=True)
2011
2011
 
2012
2012
 
2013
2013
  @patch("shrecc.database.bd.projects.set_current")
@@ -2053,7 +2053,7 @@ def test_create_database_with_network_false(
2053
2053
  consumption_profile=None,
2054
2054
  inventory_resolution=None,
2055
2055
  )
2056
- mock_db.write.assert_called_once_with(activities)
2056
+ mock_db.write.assert_called_once_with(activities, signal=True)
2057
2057
 
2058
2058
 
2059
2059
  @patch("shrecc.database.bd.projects.set_current")
@@ -2082,7 +2082,7 @@ def test_create_database_empty_activities(
2082
2082
  network="True",
2083
2083
  )
2084
2084
 
2085
- mock_db.write.assert_called_once_with({})
2085
+ mock_db.write.assert_called_once_with({}, signal=True)
2086
2086
 
2087
2087
 
2088
2088
  @patch("shrecc.database.bd.projects.set_current")
@@ -653,3 +653,50 @@ def test_get_data_resumes_only_missing_required_countries(
653
653
  assert {call.kwargs["country"] for call in get_prod_mock.call_args_list} == {"fr"}
654
654
  saved = pd.read_pickle(cache)
655
655
  assert set(saved) == {"de", "fr"}
656
+
657
+
658
+ def test_get_data_force_refresh_ignores_completed_yearly_snapshot(
659
+ monkeypatch,
660
+ tmp_path,
661
+ ):
662
+ data_dir = tmp_path / "data"
663
+ cache = data_dir / "2026" / "prod_and_trade_data_2026.pkl"
664
+ cache.parent.mkdir(parents=True)
665
+ pd.to_pickle({"de": {"production mix": pd.DataFrame()}}, cache)
666
+ cache.with_suffix(".download.json").write_text(
667
+ json.dumps({"countries": {"DE": "success"}}),
668
+ encoding="utf-8",
669
+ )
670
+ fresh_production = pd.DataFrame(
671
+ {"Solar": [1.0]},
672
+ index=pd.to_datetime(["2026-08-01 00:00"]),
673
+ )
674
+ fresh_load = pd.Series(
675
+ [2.0],
676
+ index=fresh_production.index,
677
+ )
678
+ fresh_trade = pd.DataFrame(
679
+ {"FR": [3.0]},
680
+ index=fresh_production.index,
681
+ )
682
+ monkeypatch.setattr("shrecc.energy_charts.ENERGY_CHARTS_COUNTRIES", ("DE",))
683
+ get_prod_mock = MagicMock(return_value=(fresh_production, fresh_load, ["Solar"]))
684
+ get_trade_mock = MagicMock(return_value=(fresh_trade, ["FR"]))
685
+ cleaning_mock = MagicMock(return_value="cleaned")
686
+ monkeypatch.setattr("shrecc.energy_charts.get_prod", get_prod_mock)
687
+ monkeypatch.setattr("shrecc.energy_charts.get_trade", get_trade_mock)
688
+ monkeypatch.setattr("shrecc.energy_charts.cleaning_data", cleaning_mock)
689
+
690
+ result = get_data(
691
+ 2026,
692
+ path_to_data=data_dir,
693
+ required_countries=["DE"],
694
+ request_interval=0,
695
+ force_refresh=True,
696
+ )
697
+
698
+ assert result == "cleaned"
699
+ get_prod_mock.assert_called_once()
700
+ get_trade_mock.assert_called_once()
701
+ saved = pd.read_pickle(cache)
702
+ assert saved["de"]["production mix"].equals(fresh_production)
@@ -9,6 +9,7 @@ from shrecc.pipeline import (
9
9
  NewDatabase,
10
10
  _adapt_profile_to_tyndp_times,
11
11
  )
12
+ from shrecc.result_store import CacheTimeSelectionError
12
13
 
13
14
 
14
15
  def _canonical_results(year):
@@ -331,6 +332,101 @@ def test_historical_create_repairs_cache_missing_required_country(
331
332
  assert "NL" in database.results()["consumer_country"]
332
333
 
333
334
 
335
+ def test_historical_create_refreshes_cache_missing_requested_times(
336
+ monkeypatch,
337
+ tmp_path,
338
+ ):
339
+ cache_dir = tmp_path / "2026" / "consumption_results_v1"
340
+ cache_dir.mkdir(parents=True)
341
+ (cache_dir / "manifest.json").write_text("{}", encoding="utf-8")
342
+
343
+ complete = _canonical_results(2026)
344
+ load_results = MagicMock(
345
+ side_effect=[
346
+ CacheTimeSelectionError(
347
+ "The requested time selection contains no cached results"
348
+ ),
349
+ complete,
350
+ ]
351
+ )
352
+ download = MagicMock(return_value=_energy_charts_data())
353
+ process = MagicMock(return_value=cache_dir)
354
+ monkeypatch.setattr(
355
+ "shrecc.pipeline.load_consumption_result_cache",
356
+ load_results,
357
+ )
358
+ monkeypatch.setattr("shrecc.pipeline.get_energy_charts_data", download)
359
+ monkeypatch.setattr("shrecc.pipeline.process_energy_charts_data", process)
360
+ monkeypatch.setattr("shrecc.pipeline.load_ecoinvent_mapping", MagicMock())
361
+ monkeypatch.setattr(
362
+ "shrecc.pipeline.map_consumption_mix_to_ecoinvent_activities",
363
+ MagicMock(
364
+ return_value=(
365
+ _activity_mix(2026),
366
+ xr.zeros_like(complete["consumption_mix"]),
367
+ )
368
+ ),
369
+ )
370
+ monkeypatch.setattr(
371
+ "shrecc.pipeline.activity_mix_to_database_table",
372
+ MagicMock(return_value=_database_table()),
373
+ )
374
+
375
+ NewDatabase(
376
+ years=2026,
377
+ countries=["FR"],
378
+ project_name="project",
379
+ bg_db_name="ecoinvent",
380
+ my_db_name="shrecc_FR_2026",
381
+ times=["2026-06-01 10:00"],
382
+ data_dir=tmp_path,
383
+ download=True,
384
+ ).create()
385
+
386
+ download.assert_called_once_with(
387
+ 2026,
388
+ path_to_data=tmp_path,
389
+ required_countries=("FR",),
390
+ force_refresh=True,
391
+ )
392
+ process.assert_called_once()
393
+ assert load_results.call_count == 2
394
+
395
+
396
+ def test_historical_create_does_not_refresh_times_when_download_disabled(
397
+ monkeypatch,
398
+ tmp_path,
399
+ ):
400
+ cache_dir = tmp_path / "2026" / "consumption_results_v1"
401
+ cache_dir.mkdir(parents=True)
402
+ (cache_dir / "manifest.json").write_text("{}", encoding="utf-8")
403
+ download = MagicMock()
404
+ monkeypatch.setattr(
405
+ "shrecc.pipeline.load_consumption_result_cache",
406
+ MagicMock(
407
+ side_effect=CacheTimeSelectionError(
408
+ "The requested time selection contains no cached results"
409
+ )
410
+ ),
411
+ )
412
+ monkeypatch.setattr("shrecc.pipeline.get_energy_charts_data", download)
413
+
414
+ database = NewDatabase(
415
+ years=2026,
416
+ countries=["FR"],
417
+ project_name="project",
418
+ bg_db_name="ecoinvent",
419
+ my_db_name="shrecc_FR_2026",
420
+ times=["2026-08-01 10:00"],
421
+ data_dir=tmp_path,
422
+ download=False,
423
+ )
424
+
425
+ with pytest.raises(CacheTimeSelectionError, match="download=False"):
426
+ database.create()
427
+ download.assert_not_called()
428
+
429
+
334
430
  def test_prospective_create_maps_with_premise_and_write_is_separate(monkeypatch):
335
431
  year = 2040
336
432
  times = pd.date_range(f"{year}-06-01 10:00", periods=2, freq="h")
@@ -115,6 +115,64 @@ def test_chunked_solver_matches_single_solve():
115
115
  xr.testing.assert_allclose(xr.concat(chunks, dim="time"), expected)
116
116
 
117
117
 
118
+ def test_solver_fills_missing_consumption_countries_with_zero():
119
+ """Trade-only nodes absent from consumption_volume get zero consumption.
120
+
121
+ SA-like external source nodes export into the modelled region but have
122
+ no domestic production or demand tracked in the dataset. They should be
123
+ treated as zero-consumption nodes; their exports are handled by the
124
+ unresolved_technology path in the solver.
125
+ """
126
+ time = pd.to_datetime(["2040-01-01 00:00"])
127
+ production_volume = xr.DataArray(
128
+ [[[100.0], [0.0]]],
129
+ dims=("time", "producer_country", "technology"),
130
+ coords={
131
+ "time": time,
132
+ "producer_country": ["A", "EXT"],
133
+ "technology": ["electricity"],
134
+ },
135
+ )
136
+ # EXT exports 30 MWh into A; it has no domestic generation or demand data.
137
+ trade_volume = xr.DataArray(
138
+ np.zeros((1, 2, 2)),
139
+ dims=("time", "exporter_country", "importer_country"),
140
+ coords={
141
+ "time": time,
142
+ "exporter_country": ["A", "EXT"],
143
+ "importer_country": ["A", "EXT"],
144
+ },
145
+ )
146
+ trade_volume.loc[{"exporter_country": "EXT", "importer_country": "A"}] = 30
147
+
148
+ # consumption_volume only covers A — EXT is a trade-only node.
149
+ consumption_volume = xr.DataArray(
150
+ [[125.0]],
151
+ dims=("time", "consumer_country"),
152
+ coords={"time": time, "consumer_country": ["A"]},
153
+ )
154
+
155
+ # Should not raise; EXT consumption is filled with zero (not balance,
156
+ # which would be negative and cause a ValueError).
157
+ results, _ = solve_consumption_system(
158
+ production_volume,
159
+ trade_volume,
160
+ consumption_volume=consumption_volume,
161
+ unresolved_technology="unknown",
162
+ )
163
+
164
+ # A's measured consumption is preserved.
165
+ np.testing.assert_allclose(
166
+ results["consumption_volume"].sel(consumer_country="A").values,
167
+ 125.0,
168
+ )
169
+ # EXT has zero tracked consumption.
170
+ np.testing.assert_allclose(
171
+ results["consumption_volume"].sel(consumer_country="EXT").values,
172
+ 0.0,
173
+ )
174
+
175
+
118
176
  def test_chunked_solver_rejects_cross_chunk_month_hour_imputation():
119
177
  production_volume, trade_volume = _solver_inputs()
120
178
  times = pd.date_range("2040-01-01", periods=2, freq="h")
@@ -462,3 +462,71 @@ def test_consumption_mix_zero_consumption_still_raises_by_default():
462
462
 
463
463
  with pytest.raises(ValueError, match="zero total consumption"):
464
464
  tyndp.consumption_mix_from_z_gross(Z_gross)
465
+
466
+
467
+ def test_read_excel_dataframe_falls_back_when_calamine_missing(tmp_path, monkeypatch):
468
+ """_read_excel_dataframe retries without engine when calamine is not installed."""
469
+ xlsx = tmp_path / "data.xlsx"
470
+ pd.DataFrame({"a": [1, 2]}).to_excel(xlsx, index=False)
471
+
472
+ calls = []
473
+
474
+ original_read_excel = pd.read_excel
475
+
476
+ def mock_read_excel(filename, **kwargs):
477
+ calls.append(kwargs.get("engine"))
478
+ if kwargs.get("engine") == "calamine":
479
+ raise ImportError("Missing optional dependency 'python-calamine'.")
480
+ return original_read_excel(filename, **kwargs)
481
+
482
+ monkeypatch.setattr(pd, "read_excel", mock_read_excel)
483
+
484
+ result = tyndp._read_excel_dataframe(xlsx, engine="calamine")
485
+
486
+ assert calls == ["calamine", None]
487
+ assert list(result.columns) == ["a"]
488
+
489
+
490
+ def test_read_excel_dataframe_reraises_import_error_without_engine(tmp_path, monkeypatch):
491
+ """_read_excel_dataframe does not swallow ImportErrors unrelated to engine."""
492
+ xlsx = tmp_path / "data.xlsx"
493
+ pd.DataFrame({"a": [1]}).to_excel(xlsx, index=False)
494
+
495
+ def mock_read_excel(filename, **kwargs):
496
+ raise ImportError("some other import problem")
497
+
498
+ monkeypatch.setattr(pd, "read_excel", mock_read_excel)
499
+
500
+ with pytest.raises(ImportError, match="some other import problem"):
501
+ tyndp._read_excel_dataframe(xlsx)
502
+
503
+
504
+ def test_extract_tyndp_workbook_from_zip_finds_xlsx(tmp_path):
505
+ """_extract_tyndp_workbook_from_zip falls back to .xlsx when name is not matched."""
506
+ zip_path = tmp_path / "NT2030CY2009.zip"
507
+ workbook = tmp_path / "MMStandardOutputFile_NT2030_Plexos_CY2009_2.5_v40.xlsx"
508
+
509
+ with zipfile.ZipFile(zip_path, "w") as archive:
510
+ archive.writestr(
511
+ "MMStandardOutputFile_NT2030_Plexos_CY2009_2.5_v40.xlsx",
512
+ b"xlsx-bytes",
513
+ )
514
+ # macOS metadata entry that must be ignored
515
+ archive.writestr(
516
+ "__MACOSX/._MMStandardOutputFile_NT2030_Plexos_CY2009_2.5_v40.xlsx",
517
+ b"meta",
518
+ )
519
+
520
+ tyndp._extract_tyndp_workbook_from_zip(zip_path, workbook)
521
+
522
+ assert workbook.read_bytes() == b"xlsx-bytes"
523
+
524
+
525
+ def test_parse_tyndp_datetime_index_handles_uppercase_months():
526
+ """NT workbooks use all-caps month abbreviations; parsing must normalise them."""
527
+ index = pd.Index(["01JAN00:00", "15SEP12:00", "31DEC23:00"])
528
+ result = tyndp._parse_tyndp_datetime_index(index, model_year=2030)
529
+ assert isinstance(result, pd.DatetimeIndex)
530
+ assert result[0] == pd.Timestamp("2030-01-01 00:00")
531
+ assert result[1] == pd.Timestamp("2030-09-15 12:00")
532
+ assert result[2] == pd.Timestamp("2030-12-31 23:00")
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes