wbtools 3.3.0__tar.gz → 3.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {wbtools-3.3.0/wbtools.egg-info → wbtools-3.4.0}/PKG-INFO +1 -1
  2. {wbtools-3.3.0 → wbtools-3.4.0}/pyproject.toml +1 -1
  3. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/corpus.py +12 -5
  4. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/paper.py +38 -15
  5. {wbtools-3.3.0 → wbtools-3.4.0/wbtools.egg-info}/PKG-INFO +1 -1
  6. {wbtools-3.3.0 → wbtools-3.4.0}/LICENSE +0 -0
  7. {wbtools-3.3.0 → wbtools-3.4.0}/README.md +0 -0
  8. {wbtools-3.3.0 → wbtools-3.4.0}/setup.cfg +0 -0
  9. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/__init__.py +0 -0
  10. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/__init__.py +0 -0
  11. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/abstract_manager.py +0 -0
  12. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/afp.py +0 -0
  13. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/antibody.py +0 -0
  14. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/dbmanager.py +0 -0
  15. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/expression.py +0 -0
  16. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/gene.py +0 -0
  17. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/generic.py +0 -0
  18. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/paper.py +0 -0
  19. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/db/person.py +0 -0
  20. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/__init__.py +0 -0
  21. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/email/__init__.py +0 -0
  22. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/email/generic.py +0 -0
  23. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/__init__.py +0 -0
  24. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/common.py +0 -0
  25. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/__init__.py +0 -0
  26. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/email_addresses.py +0 -0
  27. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/ntt_extractor.py +0 -0
  28. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/variations.py +0 -0
  29. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/__init__.py +0 -0
  30. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/abstract_index.py +0 -0
  31. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/textpresso.py +0 -0
  32. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_preprocessing.py +0 -0
  33. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_similarity.py +0 -0
  34. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/scraping.py +0 -0
  35. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/lib/timeout.py +0 -0
  36. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/__init__.py +0 -0
  37. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/literature/person.py +0 -0
  38. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/utils/__init__.py +0 -0
  39. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/utils/alliance.py +0 -0
  40. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools/utils/auth_utils.py +0 -0
  41. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/SOURCES.txt +0 -0
  42. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/dependency_links.txt +0 -0
  43. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/requires.txt +0 -0
  44. {wbtools-3.3.0 → wbtools-3.4.0}/wbtools.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: wbtools
3
- Version: 3.3.0
3
+ Version: 3.4.0
4
4
  Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
5
5
  Author-email: Valerio Arnaboldi <valearna@caltech.edu>
6
6
  Project-URL: Homepage, https://github.com/WormBase/wbtools
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "wbtools"
7
- version = "3.3.0"
7
+ version = "3.4.0"
8
8
  authors = [
9
9
  {name = "Valerio Arnaboldi", email = "valearna@caltech.edu"},
10
10
  ]
@@ -8,7 +8,7 @@ from wbtools.db.dbmanager import WBDBManager
8
8
  from wbtools.lib.nlp.common import PaperSections
9
9
  from wbtools.lib.nlp.text_preprocessing import preprocess
10
10
  from wbtools.lib.nlp.text_similarity import get_softcosine_index, get_similar_documents, SimilarityResult
11
- from wbtools.literature.paper import WBPaper
11
+ from wbtools.literature.paper import WBPaper, ABCRequestError
12
12
 
13
13
 
14
14
  logger = logging.getLogger(__name__)
@@ -117,10 +117,17 @@ class CorpusManager(object):
117
117
  logger.info("Loading bib info for paper")
118
118
  if paper.load_bib_info() is False:
119
119
  continue
120
- if exclude_no_author_email and not paper.get_authors_with_email_address_in_wb(
121
- blacklisted_email_addresses=blacklisted_email_addresses):
122
- logger.info("Skipping paper without any email address in text with records in WB")
123
- continue
120
+ if exclude_no_author_email:
121
+ try:
122
+ authors_with_email = paper.get_authors_with_email_address_in_wb(
123
+ blacklisted_email_addresses=blacklisted_email_addresses)
124
+ except ABCRequestError as e:
125
+ logger.warning(f"Skipping paper: {e}")
126
+ continue
127
+ if not authors_with_email:
128
+ logger.info("Skipping paper without any email address in ABC or WB authors with records "
129
+ "in WB")
130
+ continue
124
131
  if load_afp_info:
125
132
  logger.info("Loading AFP info for paper")
126
133
  paper.load_afp_info_from_db(paper_ids_no_submission=afp_no_submission_ids,
@@ -69,6 +69,10 @@ def convert_pdf_to_txt(file_path):
69
69
  return []
70
70
 
71
71
 
72
+ class ABCRequestError(Exception):
73
+ """raised when data that is required to process a paper cannot be retrieved from the ABC"""
74
+
75
+
72
76
  def get_data_from_url(url, headers=None, file_type='json'):
73
77
  try:
74
78
  response = requests.request("GET", url, headers=headers)
@@ -115,6 +119,7 @@ class WBPaper(object):
115
119
  self.afp_processed = False
116
120
  self.afp_partial_submission = False
117
121
  self.afp_contact_emails = []
122
+ self.abc_email_addresses = None
118
123
  self.db_manager = db_manager
119
124
 
120
125
  def get_corresponding_author(self) -> Union[WBAuthor, None]:
@@ -322,46 +327,64 @@ class WBPaper(object):
322
327
  def extract_all_email_addresses_from_text_and_write_to_db(self):
323
328
  self.write_email_addresses_to_db(self.extract_all_email_addresses_from_text())
324
329
 
330
+ def get_abc_email_addresses(self) -> Union[List[str], None]:
331
+ """get the email addresses associated with the paper in the ABC
332
+
333
+ Returns:
334
+ Union[List[str], None]: the list of email addresses, or None if the ABC request failed
335
+ """
336
+ if self.abc_email_addresses is None:
337
+ reference_emails_api = f"https://{ABC_API}/reference/{self.agr_curie}/emails"
338
+ headers = generate_headers(get_authentication_token())
339
+ reference_emails = get_data_from_url(reference_emails_api, headers)
340
+ if reference_emails is not None:
341
+ self.abc_email_addresses = [reference_email["email_address"] for reference_email in
342
+ reference_emails]
343
+ return self.abc_email_addresses
344
+
325
345
  def get_aut_class_value_for_datatype(self, datatype: str):
326
346
  return self.aut_class_values[datatype] if self.aut_class_values[datatype] else None
327
347
 
328
348
  def get_authors_with_email_address_in_wb(self, blacklisted_email_addresses: List[str] = None,
329
349
  first_only: bool = False) -> Union[List[Tuple[WBAuthor, str]], None]:
330
350
  """
331
- Get the first email address in the paper with a corresponding person entry in WB and return the person object
332
- and the email address found in the paper, which may be more recent than the one in WB
351
+ Get the email addresses associated with the paper in the ABC that have a corresponding person entry in WB and
352
+ return the person object and the email address from the ABC, which may be more recent than the one in WB
333
353
 
334
354
  Args:
335
355
  blacklisted_email_addresses (List[str]): a list of email addresses to be excluded from the search
336
356
  first_only (bool): whether to return only the first available author
337
357
 
338
358
  Returns:
339
- Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address found in the paper.
359
+ Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address from the ABC.
340
360
  If no email is found with a corresponding person in WB, then the function
341
361
  will return the corresponding author associated with the paper in WB and
342
362
  its email address, if any, otherwise None.
363
+
364
+ Raises:
365
+ ABCRequestError: if the email addresses cannot be retrieved from the ABC
343
366
  """
344
367
  result = []
345
- extracted_addresses = self.extract_all_email_addresses_from_text()
346
- if not extracted_addresses:
347
- extracted_addresses = self.extract_all_email_addresses_from_text(self.get_text_docs(
348
- include_supplemental=False, return_concatenated=True).replace(". ", "."))
349
- blacklisted_email_addresses = set(blacklisted_email_addresses)
350
- if extracted_addresses:
351
- for extracted_address in extracted_addresses:
352
- if "'" not in extracted_address:
368
+ abc_addresses = self.get_abc_email_addresses()
369
+ if abc_addresses is None:
370
+ raise ABCRequestError(f"Could not retrieve email addresses from ABC for paper {self.paper_id} "
371
+ f"({self.agr_curie})")
372
+ blacklisted_email_addresses = set(blacklisted_email_addresses or [])
373
+ if abc_addresses:
374
+ for abc_address in abc_addresses:
375
+ if "'" not in abc_address:
353
376
  person_id = self.db_manager.get_db_manager(
354
- WBPersonDBManager).get_person_id_from_email_address(extracted_address)
377
+ WBPersonDBManager).get_person_id_from_email_address(abc_address)
355
378
  if person_id:
356
379
  current_address = self.db_manager.get_db_manager(
357
380
  WBPersonDBManager).get_current_email_address_for_person(person_id)
358
381
  if current_address and current_address not in blacklisted_email_addresses:
359
382
  result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
360
383
  person_id=person_id), current_address))
361
- if (extracted_address not in blacklisted_email_addresses and
362
- extracted_address != current_address):
384
+ if (abc_address not in blacklisted_email_addresses and
385
+ abc_address != current_address):
363
386
  result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
364
- person_id=person_id), extracted_address))
387
+ person_id=person_id), abc_address))
365
388
  if first_only and result:
366
389
  return result
367
390
  if not result:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: wbtools
3
- Version: 3.3.0
3
+ Version: 3.4.0
4
4
  Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
5
5
  Author-email: Valerio Arnaboldi <valearna@caltech.edu>
6
6
  Project-URL: Homepage, https://github.com/WormBase/wbtools
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes