wbtools 3.3.0__tar.gz → 3.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wbtools-3.3.0/wbtools.egg-info → wbtools-3.4.0}/PKG-INFO +1 -1
- {wbtools-3.3.0 → wbtools-3.4.0}/pyproject.toml +1 -1
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/corpus.py +12 -5
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/paper.py +38 -15
- {wbtools-3.3.0 → wbtools-3.4.0/wbtools.egg-info}/PKG-INFO +1 -1
- {wbtools-3.3.0 → wbtools-3.4.0}/LICENSE +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/README.md +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/setup.cfg +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/abstract_manager.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/afp.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/antibody.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/dbmanager.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/expression.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/gene.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/generic.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/paper.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/person.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/email/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/email/generic.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/common.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/email_addresses.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/ntt_extractor.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/variations.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/abstract_index.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/textpresso.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_preprocessing.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_similarity.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/scraping.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/timeout.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/person.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/utils/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/utils/alliance.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/utils/auth_utils.py +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/SOURCES.txt +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/dependency_links.txt +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/requires.txt +0 -0
- {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: wbtools
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.4.0
|
|
4
4
|
Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
|
|
5
5
|
Author-email: Valerio Arnaboldi <valearna@caltech.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/WormBase/wbtools
|
|
@@ -8,7 +8,7 @@ from wbtools.db.dbmanager import WBDBManager
|
|
|
8
8
|
from wbtools.lib.nlp.common import PaperSections
|
|
9
9
|
from wbtools.lib.nlp.text_preprocessing import preprocess
|
|
10
10
|
from wbtools.lib.nlp.text_similarity import get_softcosine_index, get_similar_documents, SimilarityResult
|
|
11
|
-
from wbtools.literature.paper import WBPaper
|
|
11
|
+
from wbtools.literature.paper import WBPaper, ABCRequestError
|
|
12
12
|
|
|
13
13
|
|
|
14
14
|
logger = logging.getLogger(__name__)
|
|
@@ -117,10 +117,17 @@ class CorpusManager(object):
|
|
|
117
117
|
logger.info("Loading bib info for paper")
|
|
118
118
|
if paper.load_bib_info() is False:
|
|
119
119
|
continue
|
|
120
|
-
if exclude_no_author_email
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
120
|
+
if exclude_no_author_email:
|
|
121
|
+
try:
|
|
122
|
+
authors_with_email = paper.get_authors_with_email_address_in_wb(
|
|
123
|
+
blacklisted_email_addresses=blacklisted_email_addresses)
|
|
124
|
+
except ABCRequestError as e:
|
|
125
|
+
logger.warning(f"Skipping paper: {e}")
|
|
126
|
+
continue
|
|
127
|
+
if not authors_with_email:
|
|
128
|
+
logger.info("Skipping paper without any email address in ABC or WB authors with records "
|
|
129
|
+
"in WB")
|
|
130
|
+
continue
|
|
124
131
|
if load_afp_info:
|
|
125
132
|
logger.info("Loading AFP info for paper")
|
|
126
133
|
paper.load_afp_info_from_db(paper_ids_no_submission=afp_no_submission_ids,
|
|
@@ -69,6 +69,10 @@ def convert_pdf_to_txt(file_path):
|
|
|
69
69
|
return []
|
|
70
70
|
|
|
71
71
|
|
|
72
|
+
class ABCRequestError(Exception):
|
|
73
|
+
"""raised when data that is required to process a paper cannot be retrieved from the ABC"""
|
|
74
|
+
|
|
75
|
+
|
|
72
76
|
def get_data_from_url(url, headers=None, file_type='json'):
|
|
73
77
|
try:
|
|
74
78
|
response = requests.request("GET", url, headers=headers)
|
|
@@ -115,6 +119,7 @@ class WBPaper(object):
|
|
|
115
119
|
self.afp_processed = False
|
|
116
120
|
self.afp_partial_submission = False
|
|
117
121
|
self.afp_contact_emails = []
|
|
122
|
+
self.abc_email_addresses = None
|
|
118
123
|
self.db_manager = db_manager
|
|
119
124
|
|
|
120
125
|
def get_corresponding_author(self) -> Union[WBAuthor, None]:
|
|
@@ -322,46 +327,64 @@ class WBPaper(object):
|
|
|
322
327
|
def extract_all_email_addresses_from_text_and_write_to_db(self):
|
|
323
328
|
self.write_email_addresses_to_db(self.extract_all_email_addresses_from_text())
|
|
324
329
|
|
|
330
|
+
def get_abc_email_addresses(self) -> Union[List[str], None]:
|
|
331
|
+
"""get the email addresses associated with the paper in the ABC
|
|
332
|
+
|
|
333
|
+
Returns:
|
|
334
|
+
Union[List[str], None]: the list of email addresses, or None if the ABC request failed
|
|
335
|
+
"""
|
|
336
|
+
if self.abc_email_addresses is None:
|
|
337
|
+
reference_emails_api = f"https://{ABC_API}/reference/{self.agr_curie}/emails"
|
|
338
|
+
headers = generate_headers(get_authentication_token())
|
|
339
|
+
reference_emails = get_data_from_url(reference_emails_api, headers)
|
|
340
|
+
if reference_emails is not None:
|
|
341
|
+
self.abc_email_addresses = [reference_email["email_address"] for reference_email in
|
|
342
|
+
reference_emails]
|
|
343
|
+
return self.abc_email_addresses
|
|
344
|
+
|
|
325
345
|
def get_aut_class_value_for_datatype(self, datatype: str):
|
|
326
346
|
return self.aut_class_values[datatype] if self.aut_class_values[datatype] else None
|
|
327
347
|
|
|
328
348
|
def get_authors_with_email_address_in_wb(self, blacklisted_email_addresses: List[str] = None,
|
|
329
349
|
first_only: bool = False) -> Union[List[Tuple[WBAuthor, str]], None]:
|
|
330
350
|
"""
|
|
331
|
-
Get the
|
|
332
|
-
and the email address
|
|
351
|
+
Get the email addresses associated with the paper in the ABC that have a corresponding person entry in WB and
|
|
352
|
+
return the person object and the email address from the ABC, which may be more recent than the one in WB
|
|
333
353
|
|
|
334
354
|
Args:
|
|
335
355
|
blacklisted_email_addresses (List[str]): a list of email addresses to be excluded from the search
|
|
336
356
|
first_only (bool): whether to return only the first available author
|
|
337
357
|
|
|
338
358
|
Returns:
|
|
339
|
-
Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address
|
|
359
|
+
Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address from the ABC.
|
|
340
360
|
If no email is found with a corresponding person in WB, then the function
|
|
341
361
|
will return the corresponding author associated with the paper in WB and
|
|
342
362
|
its email address, if any, otherwise None.
|
|
363
|
+
|
|
364
|
+
Raises:
|
|
365
|
+
ABCRequestError: if the email addresses cannot be retrieved from the ABC
|
|
343
366
|
"""
|
|
344
367
|
result = []
|
|
345
|
-
|
|
346
|
-
if
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
blacklisted_email_addresses = set(blacklisted_email_addresses)
|
|
350
|
-
if
|
|
351
|
-
for
|
|
352
|
-
if "'" not in
|
|
368
|
+
abc_addresses = self.get_abc_email_addresses()
|
|
369
|
+
if abc_addresses is None:
|
|
370
|
+
raise ABCRequestError(f"Could not retrieve email addresses from ABC for paper {self.paper_id} "
|
|
371
|
+
f"({self.agr_curie})")
|
|
372
|
+
blacklisted_email_addresses = set(blacklisted_email_addresses or [])
|
|
373
|
+
if abc_addresses:
|
|
374
|
+
for abc_address in abc_addresses:
|
|
375
|
+
if "'" not in abc_address:
|
|
353
376
|
person_id = self.db_manager.get_db_manager(
|
|
354
|
-
WBPersonDBManager).get_person_id_from_email_address(
|
|
377
|
+
WBPersonDBManager).get_person_id_from_email_address(abc_address)
|
|
355
378
|
if person_id:
|
|
356
379
|
current_address = self.db_manager.get_db_manager(
|
|
357
380
|
WBPersonDBManager).get_current_email_address_for_person(person_id)
|
|
358
381
|
if current_address and current_address not in blacklisted_email_addresses:
|
|
359
382
|
result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
|
|
360
383
|
person_id=person_id), current_address))
|
|
361
|
-
if (
|
|
362
|
-
|
|
384
|
+
if (abc_address not in blacklisted_email_addresses and
|
|
385
|
+
abc_address != current_address):
|
|
363
386
|
result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
|
|
364
|
-
person_id=person_id),
|
|
387
|
+
person_id=person_id), abc_address))
|
|
365
388
|
if first_only and result:
|
|
366
389
|
return result
|
|
367
390
|
if not result:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: wbtools
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.4.0
|
|
4
4
|
Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
|
|
5
5
|
Author-email: Valerio Arnaboldi <valearna@caltech.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/WormBase/wbtools
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|