python-materialsdb 0.3.0__py3-none-any.whl → 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
materialsdb/cache.py CHANGED
@@ -2,6 +2,7 @@ import os
2
2
  import pathlib
3
3
  import urllib.request
4
4
  from collections import namedtuple
5
+ from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
5
6
 
6
7
  from lxml import etree
7
8
 
@@ -85,25 +86,77 @@ def _plan_index(index):
85
86
  return _IndexPlan(index, new_index, unchanged, todos)
86
87
 
87
88
 
88
- def update_producers_data(url_list=MATERIALSDBINDEXURLLIST, on_progress=None):
89
- """on_progress(done, total, name) is called after each producer download
90
- with a global running total; returning a truthy value stops the update
91
- between two downloads (the partial Report is returned as-is)."""
89
+ def update_producers_data(url_list=MATERIALSDBINDEXURLLIST, on_progress=None, max_workers: int = 8):
90
+ """Update the cached producer files from the index URLs.
91
+
92
+ Downloads run on a bounded thread pool (max_workers). on_progress(done,
93
+ total, name) is called once per completed download from the calling
94
+ thread with a monotonic running total; returning a truthy value stops
95
+ NEW submissions (queued jobs are abandoned, in-flight ones finish but
96
+ no longer count toward the partial Report). A download error stops the
97
+ run and the first exception is re-raised after the pool drains.
98
+ """
92
99
  plans = [_plan_index(index) for index in url_list]
93
- total = sum(len(plan.todos) for plan in plans)
94
- existing, updated, deleted = [], [], []
100
+ jobs = [
101
+ (plan, company, cached_producer, producer_path)
102
+ for plan in plans
103
+ for (company, cached_producer, producer_path) in plan.todos
104
+ ]
105
+ total = len(jobs)
106
+ workers = max(1, min(int(max_workers), total or 1))
107
+
108
+ existing: list = []
109
+ updated: list = []
110
+ deleted: list = []
95
111
  done = 0
112
+ cancel_requested = False
113
+ failure = None
114
+
115
+ def _download(job):
116
+ _, company, cached_producer, producer_path = job
117
+ deleted_path = None
118
+ if cached_producer is not None:
119
+ deleted_path = get_producers_dir() / pathlib.Path(cached_producer.get("href")).name
120
+ deleted_path.unlink(True)
121
+ urllib.request.urlretrieve(company.get("href"), producer_path)
122
+ return deleted_path
123
+
124
+ pool = ThreadPoolExecutor(max_workers=workers)
125
+ pending = list(jobs) # pops from the FRONT, submission order preserved
126
+ futures: dict = {}
127
+ try:
128
+ while futures or pending:
129
+ while pending and len(futures) < workers:
130
+ job = pending.pop(0)
131
+ futures[pool.submit(_download, job)] = job
132
+ if not futures:
133
+ break
134
+ ready, _ = wait(set(futures), return_when=FIRST_COMPLETED)
135
+ for future in ready:
136
+ job = futures.pop(future)
137
+ try:
138
+ deleted_path = future.result()
139
+ except Exception as err: # noqa: BLE001 - re-raised below, caller decides
140
+ if failure is None:
141
+ failure = err
142
+ break
143
+ if deleted_path is not None:
144
+ deleted.append(deleted_path)
145
+ updated.append(job[3])
146
+ done += 1
147
+ if on_progress is not None and on_progress(done, total, pathlib.Path(job[3]).name):
148
+ cancel_requested = True
149
+ break
150
+ if cancel_requested or failure is not None:
151
+ break
152
+ finally:
153
+ pool.shutdown(wait=True, cancel_futures=True)
154
+
155
+ if failure is not None:
156
+ raise failure
157
+ if cancel_requested:
158
+ return Report(existing, updated, deleted)
96
159
  for plan in plans:
97
- for company, cached_producer, producer_path in plan.todos:
98
- if cached_producer is not None:
99
- cached_path = get_producers_dir() / pathlib.Path(cached_producer.get("href")).name
100
- deleted.append(cached_path)
101
- cached_path.unlink(True)
102
- urllib.request.urlretrieve(company.get("href"), producer_path)
103
- updated.append(producer_path)
104
- done += 1
105
- if on_progress is not None and on_progress(done, total, pathlib.Path(company.get("href")).name):
106
- return Report(existing, updated, deleted)
107
160
  if plan.todos:
108
161
  plan.new_index.write(str(get_cached_index_path(plan.index)))
109
162
  existing.extend(plan.unchanged)
materialsdb/serialiser.py CHANGED
@@ -25,9 +25,31 @@ def get_xml_schema() -> str:
25
25
  return str(Path(__file__).parent / "schema/materialsdb103.xsd")
26
26
 
27
27
 
28
+ @cache
29
+ def cached_type_hints(cls) -> dict:
30
+ return typing.get_type_hints(cls)
31
+
32
+
33
+ @cache
34
+ def _tag_local_name(tag: str) -> str:
35
+ m = re.search("{.*}(.*)", tag)
36
+ return m.group(1) if m else tag
37
+
38
+
28
39
  def get_element_name(element: objectify.ObjectifiedElement) -> str:
29
- m = re.search("{.*}(.*)", element.tag)
30
- return m.group(1) if m else element.tag
40
+ return _tag_local_name(element.tag)
41
+
42
+
43
+ @cache
44
+ def _is_optional(type_hint):
45
+ return typing.get_origin(type_hint) is typing.Union and typing.get_args(type_hint)[1] is type(None)
46
+
47
+
48
+ @cache
49
+ def _strip_optional(type_hint):
50
+ if _is_optional(type_hint):
51
+ return typing.get_args(type_hint)[0]
52
+ return type_hint
31
53
 
32
54
 
33
55
  def create_element_maker():
@@ -49,11 +71,6 @@ def get_valid_root(tree: objectify.ObjectifiedElement) -> objectify.ObjectifiedE
49
71
  return root
50
72
 
51
73
 
52
- @cache
53
- def cached_type_hints(cls) -> dict:
54
- return typing.get_type_hints(cls)
55
-
56
-
57
74
  class XmlDeserialiser:
58
75
  def __init__(self):
59
76
  self.schema = etree.XMLSchema(file=get_xml_schema())
@@ -93,7 +110,7 @@ class XmlDeserialiser:
93
110
  value = element.get(attrib)
94
111
  if value is None:
95
112
  continue
96
- base_class = self.strip_optional(type_hints[attrib])
113
+ base_class = _strip_optional(type_hints[attrib])
97
114
 
98
115
  kwargs[attrib] = base_class(value)
99
116
  for child_name in getattr(element_class, "xml_elements", ()):
@@ -127,12 +144,10 @@ class XmlDeserialiser:
127
144
  return instance
128
145
 
129
146
  def strip_optional(self, type_hint):
130
- if self.is_optional(type_hint):
131
- return typing.get_args(type_hint)[0]
132
- return type_hint
147
+ return _strip_optional(type_hint)
133
148
 
134
149
  def is_optional(self, type_hint):
135
- return typing.get_origin(type_hint) is typing.Union and typing.get_args(type_hint)[1] is type(None)
150
+ return _is_optional(type_hint)
136
151
 
137
152
  def cls_name(self, name: str) -> str:
138
153
  return name if name[0].isupper() else name.title()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: python-materialsdb
3
- Version: 0.3.0
3
+ Version: 0.3.1
4
4
  Summary: A library to work with materialsdb.org open standard for building materials.
5
5
  Author-email: Cyril Waechter <cyrwae@hotmail.com>
6
6
  License-Expression: GPL-3.0-or-later
@@ -1,10 +1,10 @@
1
1
  materialsdb/__init__.py,sha256=lml7ffWFF3-0MD-Z7YbPZNireUpDO7tR9B3KNamrI0k,127
2
- materialsdb/cache.py,sha256=9XpJ7I5mLNpQZsVZSggIdUwS3votombJ9PfjbVbl-OA,4751
2
+ materialsdb/cache.py,sha256=wPCksig6hidUssjRx_KaPk9V5ZeWbHfbs0Qt2XTuLPk,6594
3
3
  materialsdb/classes.py,sha256=WdXNiCqFcevgWrqDZKLR8HPio3BwcmOJ1NyVGArQ0hw,21549
4
4
  materialsdb/config.py,sha256=Tp14G_Y_XEkYshwrp4gd9XjGb5e_mO-hxYOevCTVnmI,1291
5
5
  materialsdb/construction.py,sha256=29H5gt3EX0LFNJ7ftE0Ob7huQgyZzx3ygW7nNcUSo5s,17817
6
6
  materialsdb/query.py,sha256=h9kP3Zc41l4h9bcxjwp-UUb8ezsG0A3ZkraJjaQZVVI,620
7
- materialsdb/serialiser.py,sha256=GL_e1FW6inuvfLvqQzp81L8XgoByRtVtd7nWtrYKWmY,7056
7
+ materialsdb/serialiser.py,sha256=J6bUc7eKYD569TrOWVAK5ZDeb7cdD2YbF0nD6jf3iPY,7263
8
8
  materialsdb/store.py,sha256=_Iu425X7-sYE7A2vvQVrR1R6ju2R63m_BK43c9y3mhM,11000
9
9
  materialsdb/summary.py,sha256=vl6LhVIfN4KHN-KZxU6oVBYut0FPA3E_baGp_fegnBU,2558
10
10
  materialsdb/utils.py,sha256=2xGwr6b4NgOtjT5lxpFVN9jnOSlDsmZPnelSGEs5MDs,3068
@@ -25,9 +25,9 @@ materialsdb/ifc/project_library.py,sha256=hOvirB73ZiTYiP6rOu6eN1dU2ZXFRgSOGpQrH2
25
25
  materialsdb/schema/MaterialsDBIndex100.xsd,sha256=oK81n2b3_82Mt5G31yEm9vwkMjUmNyM6jglt_zpUQZc,2199
26
26
  materialsdb/schema/materialsdb102.xsd,sha256=IaC6E_B3N1VaU9lhrl6ZGwi8m2rk2p7HTEZKN_Yj2AY,32880
27
27
  materialsdb/schema/materialsdb103.xsd,sha256=iX7Q1ejWEN_ZSuRVCX40zjYt5UraY3f_JxBzqZUFu7Q,37762
28
- python_materialsdb-0.3.0.dist-info/licenses/LICENSE.md,sha256=M7wm1EmMGDtwPRdg7kW4d00h1uAXjKOT3HFScYQMeiE,34916
29
- python_materialsdb-0.3.0.dist-info/METADATA,sha256=dugVsiclmutXpwQj_CI8tkEe0oGNus2Amc8vnZqit_w,7673
30
- python_materialsdb-0.3.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
31
- python_materialsdb-0.3.0.dist-info/entry_points.txt,sha256=7bCySVv6oV9hiva7Fn5Dm9oboPRIC6DU6DRlsjCLO6k,66
32
- python_materialsdb-0.3.0.dist-info/top_level.txt,sha256=5ZHbF8Oj1W24bifG65Fl3hCaivZpkO6EyOHHaRavZxE,12
33
- python_materialsdb-0.3.0.dist-info/RECORD,,
28
+ python_materialsdb-0.3.1.dist-info/licenses/LICENSE.md,sha256=M7wm1EmMGDtwPRdg7kW4d00h1uAXjKOT3HFScYQMeiE,34916
29
+ python_materialsdb-0.3.1.dist-info/METADATA,sha256=y09iI0wUVkcDhSEnA3YwIEB4S-CXQx4Ms9IIfMiYvrU,7673
30
+ python_materialsdb-0.3.1.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
31
+ python_materialsdb-0.3.1.dist-info/entry_points.txt,sha256=7bCySVv6oV9hiva7Fn5Dm9oboPRIC6DU6DRlsjCLO6k,66
32
+ python_materialsdb-0.3.1.dist-info/top_level.txt,sha256=5ZHbF8Oj1W24bifG65Fl3hCaivZpkO6EyOHHaRavZxE,12
33
+ python_materialsdb-0.3.1.dist-info/RECORD,,