wbtools 3.2.0__tar.gz → 3.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {wbtools-3.2.0/wbtools.egg-info → wbtools-3.4.0}/PKG-INFO +2 -1
  2. {wbtools-3.2.0 → wbtools-3.4.0}/pyproject.toml +2 -1
  3. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/corpus.py +12 -5
  4. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/paper.py +43 -17
  5. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/utils/alliance.py +1 -1
  6. wbtools-3.4.0/wbtools/utils/auth_utils.py +21 -0
  7. {wbtools-3.2.0 → wbtools-3.4.0/wbtools.egg-info}/PKG-INFO +2 -1
  8. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/SOURCES.txt +1 -1
  9. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/requires.txt +1 -0
  10. wbtools-3.2.0/wbtools/utils/okta_utils.py +0 -67
  11. {wbtools-3.2.0 → wbtools-3.4.0}/LICENSE +0 -0
  12. {wbtools-3.2.0 → wbtools-3.4.0}/README.md +0 -0
  13. {wbtools-3.2.0 → wbtools-3.4.0}/setup.cfg +0 -0
  14. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/__init__.py +0 -0
  15. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/__init__.py +0 -0
  16. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/abstract_manager.py +0 -0
  17. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/afp.py +0 -0
  18. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/antibody.py +0 -0
  19. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/dbmanager.py +0 -0
  20. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/expression.py +0 -0
  21. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/gene.py +0 -0
  22. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/generic.py +0 -0
  23. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/paper.py +0 -0
  24. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/person.py +0 -0
  25. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/__init__.py +0 -0
  26. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/email/__init__.py +0 -0
  27. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/email/generic.py +0 -0
  28. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/__init__.py +0 -0
  29. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/common.py +0 -0
  30. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/__init__.py +0 -0
  31. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/email_addresses.py +0 -0
  32. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/ntt_extractor.py +0 -0
  33. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/variations.py +0 -0
  34. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/__init__.py +0 -0
  35. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/abstract_index.py +0 -0
  36. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/textpresso.py +0 -0
  37. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_preprocessing.py +0 -0
  38. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_similarity.py +0 -0
  39. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/scraping.py +0 -0
  40. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/timeout.py +0 -0
  41. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/__init__.py +0 -0
  42. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/person.py +0 -0
  43. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/utils/__init__.py +0 -0
  44. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/dependency_links.txt +0 -0
  45. {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: wbtools
3
- Version: 3.2.0
3
+ Version: 3.4.0
4
4
  Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
5
5
  Author-email: Valerio Arnaboldi <valearna@caltech.edu>
6
6
  Project-URL: Homepage, https://github.com/WormBase/wbtools
@@ -24,6 +24,7 @@ Requires-Dist: requests~=2.31.0
24
24
  Requires-Dist: python-dateutil~=2.8.2
25
25
  Requires-Dist: python-dotenv~=1.0.0
26
26
  Requires-Dist: grobid-client
27
+ Requires-Dist: agr-cognito-py
27
28
 
28
29
  # WBtools
29
30
  > Interface to WormBase curation database and Text Mining functions
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "wbtools"
7
- version = "3.2.0"
7
+ version = "3.4.0"
8
8
  authors = [
9
9
  {name = "Valerio Arnaboldi", email = "valearna@caltech.edu"},
10
10
  ]
@@ -30,6 +30,7 @@ dependencies = [
30
30
  "python-dateutil~=2.8.2",
31
31
  "python-dotenv~=1.0.0",
32
32
  "grobid-client",
33
+ "agr-cognito-py",
33
34
  ]
34
35
 
35
36
  [project.urls]
@@ -8,7 +8,7 @@ from wbtools.db.dbmanager import WBDBManager
8
8
  from wbtools.lib.nlp.common import PaperSections
9
9
  from wbtools.lib.nlp.text_preprocessing import preprocess
10
10
  from wbtools.lib.nlp.text_similarity import get_softcosine_index, get_similar_documents, SimilarityResult
11
- from wbtools.literature.paper import WBPaper
11
+ from wbtools.literature.paper import WBPaper, ABCRequestError
12
12
 
13
13
 
14
14
  logger = logging.getLogger(__name__)
@@ -117,10 +117,17 @@ class CorpusManager(object):
117
117
  logger.info("Loading bib info for paper")
118
118
  if paper.load_bib_info() is False:
119
119
  continue
120
- if exclude_no_author_email and not paper.get_authors_with_email_address_in_wb(
121
- blacklisted_email_addresses=blacklisted_email_addresses):
122
- logger.info("Skipping paper without any email address in text with records in WB")
123
- continue
120
+ if exclude_no_author_email:
121
+ try:
122
+ authors_with_email = paper.get_authors_with_email_address_in_wb(
123
+ blacklisted_email_addresses=blacklisted_email_addresses)
124
+ except ABCRequestError as e:
125
+ logger.warning(f"Skipping paper: {e}")
126
+ continue
127
+ if not authors_with_email:
128
+ logger.info("Skipping paper without any email address in ABC or WB authors with records "
129
+ "in WB")
130
+ continue
124
131
  if load_afp_info:
125
132
  logger.info("Loading AFP info for paper")
126
133
  paper.load_afp_info_from_db(paper_ids_no_submission=afp_no_submission_ids,
@@ -26,7 +26,7 @@ from wbtools.lib.nlp.entity_extraction.email_addresses import get_email_addresse
26
26
  from wbtools.lib.nlp.text_preprocessing import preprocess, get_documents_from_text, PaperSections
27
27
  from wbtools.lib.timeout import timeout
28
28
  from wbtools.literature.person import WBAuthor
29
- from wbtools.utils.okta_utils import get_authentication_token, generate_headers
29
+ from wbtools.utils.auth_utils import get_authentication_token, generate_headers
30
30
 
31
31
  logger = logging.getLogger(__name__)
32
32
 
@@ -69,6 +69,10 @@ def convert_pdf_to_txt(file_path):
69
69
  return []
70
70
 
71
71
 
72
+ class ABCRequestError(Exception):
73
+ """raised when data that is required to process a paper cannot be retrieved from the ABC"""
74
+
75
+
72
76
  def get_data_from_url(url, headers=None, file_type='json'):
73
77
  try:
74
78
  response = requests.request("GET", url, headers=headers)
@@ -115,6 +119,7 @@ class WBPaper(object):
115
119
  self.afp_processed = False
116
120
  self.afp_partial_submission = False
117
121
  self.afp_contact_emails = []
122
+ self.abc_email_addresses = None
118
123
  self.db_manager = db_manager
119
124
 
120
125
  def get_corresponding_author(self) -> Union[WBAuthor, None]:
@@ -193,6 +198,7 @@ class WBPaper(object):
193
198
  blue_api_base_url = os.environ.get('API_SERVER', "literature-rest.alliancegenome.org")
194
199
  all_reffiles_for_pap_api = f'https://{blue_api_base_url}/reference/referencefile/show_all/{self.agr_curie}'
195
200
  request = urllib.request.Request(url=all_reffiles_for_pap_api)
201
+ request.add_header("Authorization", f"Bearer {get_authentication_token()}")
196
202
  request.add_header("Content-type", "application/json")
197
203
  request.add_header("Accept", "application/json")
198
204
  added_ref_files = 0
@@ -227,7 +233,9 @@ class WBPaper(object):
227
233
  raise Exception("PaperDBManager not set")
228
234
  # Get Alliance reference info from WBPaperID
229
235
  ref_info_from_xref_api = f"https://{ABC_API}/reference/by_cross_reference/WB:WBPaper{self.paper_id}"
230
- ref_info: dict = get_data_from_url(ref_info_from_xref_api)
236
+ token = get_authentication_token()
237
+ headers = generate_headers(token)
238
+ ref_info: dict = get_data_from_url(ref_info_from_xref_api, headers)
231
239
  if ref_info:
232
240
  self.abstract = ref_info["abstract"]
233
241
  self.title = ref_info["title"]
@@ -319,46 +327,64 @@ class WBPaper(object):
319
327
  def extract_all_email_addresses_from_text_and_write_to_db(self):
320
328
  self.write_email_addresses_to_db(self.extract_all_email_addresses_from_text())
321
329
 
330
+ def get_abc_email_addresses(self) -> Union[List[str], None]:
331
+ """get the email addresses associated with the paper in the ABC
332
+
333
+ Returns:
334
+ Union[List[str], None]: the list of email addresses, or None if the ABC request failed
335
+ """
336
+ if self.abc_email_addresses is None:
337
+ reference_emails_api = f"https://{ABC_API}/reference/{self.agr_curie}/emails"
338
+ headers = generate_headers(get_authentication_token())
339
+ reference_emails = get_data_from_url(reference_emails_api, headers)
340
+ if reference_emails is not None:
341
+ self.abc_email_addresses = [reference_email["email_address"] for reference_email in
342
+ reference_emails]
343
+ return self.abc_email_addresses
344
+
322
345
  def get_aut_class_value_for_datatype(self, datatype: str):
323
346
  return self.aut_class_values[datatype] if self.aut_class_values[datatype] else None
324
347
 
325
348
  def get_authors_with_email_address_in_wb(self, blacklisted_email_addresses: List[str] = None,
326
349
  first_only: bool = False) -> Union[List[Tuple[WBAuthor, str]], None]:
327
350
  """
328
- Get the first email address in the paper with a corresponding person entry in WB and return the person object
329
- and the email address found in the paper, which may be more recent than the one in WB
351
+ Get the email addresses associated with the paper in the ABC that have a corresponding person entry in WB and
352
+ return the person object and the email address from the ABC, which may be more recent than the one in WB
330
353
 
331
354
  Args:
332
355
  blacklisted_email_addresses (List[str]): a list of email addresses to be excluded from the search
333
356
  first_only (bool): whether to return only the first available author
334
357
 
335
358
  Returns:
336
- Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address found in the paper.
359
+ Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address from the ABC.
337
360
  If no email is found with a corresponding person in WB, then the function
338
361
  will return the corresponding author associated with the paper in WB and
339
362
  its email address, if any, otherwise None.
363
+
364
+ Raises:
365
+ ABCRequestError: if the email addresses cannot be retrieved from the ABC
340
366
  """
341
367
  result = []
342
- extracted_addresses = self.extract_all_email_addresses_from_text()
343
- if not extracted_addresses:
344
- extracted_addresses = self.extract_all_email_addresses_from_text(self.get_text_docs(
345
- include_supplemental=False, return_concatenated=True).replace(". ", "."))
346
- blacklisted_email_addresses = set(blacklisted_email_addresses)
347
- if extracted_addresses:
348
- for extracted_address in extracted_addresses:
349
- if "'" not in extracted_address:
368
+ abc_addresses = self.get_abc_email_addresses()
369
+ if abc_addresses is None:
370
+ raise ABCRequestError(f"Could not retrieve email addresses from ABC for paper {self.paper_id} "
371
+ f"({self.agr_curie})")
372
+ blacklisted_email_addresses = set(blacklisted_email_addresses or [])
373
+ if abc_addresses:
374
+ for abc_address in abc_addresses:
375
+ if "'" not in abc_address:
350
376
  person_id = self.db_manager.get_db_manager(
351
- WBPersonDBManager).get_person_id_from_email_address(extracted_address)
377
+ WBPersonDBManager).get_person_id_from_email_address(abc_address)
352
378
  if person_id:
353
379
  current_address = self.db_manager.get_db_manager(
354
380
  WBPersonDBManager).get_current_email_address_for_person(person_id)
355
381
  if current_address and current_address not in blacklisted_email_addresses:
356
382
  result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
357
383
  person_id=person_id), current_address))
358
- if (extracted_address not in blacklisted_email_addresses and
359
- extracted_address != current_address):
384
+ if (abc_address not in blacklisted_email_addresses and
385
+ abc_address != current_address):
360
386
  result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
361
- person_id=person_id), extracted_address))
387
+ person_id=person_id), abc_address))
362
388
  if first_only and result:
363
389
  return result
364
390
  if not result:
@@ -3,7 +3,7 @@ import logging
3
3
  import urllib.request
4
4
 
5
5
  from wbtools.lib.nlp.common import EntityType
6
- from wbtools.utils.okta_utils import get_authentication_token
6
+ from wbtools.utils.auth_utils import get_authentication_token
7
7
 
8
8
 
9
9
  logger = logging.getLogger(__name__)
@@ -0,0 +1,21 @@
1
+ """
2
+ Authentication utilities for AGR API access.
3
+
4
+ This module provides authentication token generation using AWS Cognito
5
+ via the agr_cognito_py library. The functions are re-exported for
6
+ backward compatibility with code that was using the old Okta-based
7
+ authentication.
8
+
9
+ Required environment variables for Cognito authentication:
10
+ - COGNITO_ADMIN_CLIENT_ID
11
+ - COGNITO_ADMIN_CLIENT_SECRET
12
+ - COGNITO_TOKEN_URL
13
+ - COGNITO_ADMIN_SCOPE
14
+ """
15
+
16
+ from agr_cognito_py import (
17
+ get_admin_token as get_authentication_token,
18
+ generate_headers
19
+ )
20
+
21
+ __all__ = ['get_authentication_token', 'generate_headers']
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: wbtools
3
- Version: 3.2.0
3
+ Version: 3.4.0
4
4
  Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
5
5
  Author-email: Valerio Arnaboldi <valearna@caltech.edu>
6
6
  Project-URL: Homepage, https://github.com/WormBase/wbtools
@@ -24,6 +24,7 @@ Requires-Dist: requests~=2.31.0
24
24
  Requires-Dist: python-dateutil~=2.8.2
25
25
  Requires-Dist: python-dotenv~=1.0.0
26
26
  Requires-Dist: grobid-client
27
+ Requires-Dist: agr-cognito-py
27
28
 
28
29
  # WBtools
29
30
  > Interface to WormBase curation database and Text Mining functions
@@ -39,4 +39,4 @@ wbtools/literature/paper.py
39
39
  wbtools/literature/person.py
40
40
  wbtools/utils/__init__.py
41
41
  wbtools/utils/alliance.py
42
- wbtools/utils/okta_utils.py
42
+ wbtools/utils/auth_utils.py
@@ -11,3 +11,4 @@ requests~=2.31.0
11
11
  python-dateutil~=2.8.2
12
12
  python-dotenv~=1.0.0
13
13
  grobid-client
14
+ agr-cognito-py
@@ -1,67 +0,0 @@
1
- import os
2
- from json import loads
3
- from os import environ, path
4
- import logging
5
- import logging.config
6
- from datetime import datetime
7
- from dateutil.relativedelta import relativedelta
8
- import requests
9
-
10
-
11
- def generate_headers(token):
12
- """
13
-
14
- :param token:
15
- :return:
16
- """
17
-
18
- authorization = 'Bearer ' + token
19
- headers = {
20
- 'Authorization': authorization,
21
- 'Content-Type': 'application/json',
22
- 'Accept': 'application/json'
23
- }
24
- return headers
25
-
26
-
27
- def update_okta_token(): # pragma: no cover
28
- """
29
-
30
- :return:
31
- """
32
-
33
- url = f"https://{os.environ.get('OKTA_DOMAIN')}/v1/token"
34
- headers = {
35
- 'Content-Type': 'application/x-www-form-urlencoded',
36
- 'Accept': 'application/json'
37
- }
38
- data_dict = dict()
39
- data_dict['grant_type'] = 'client_credentials'
40
- data_dict['client_id'] = environ.get('OKTA_CLIENT_ID')
41
- data_dict['client_secret'] = environ.get('OKTA_CLIENT_SECRET')
42
- data_dict['scope'] = 'admin'
43
- post_return = requests.post(url, headers=headers, data=data_dict)
44
- logging.warning(post_return.text)
45
- response_dict = loads(post_return.text)
46
- token = response_dict['access_token']
47
- okta_file = environ.get('TMP_PATH') + 'okta_token'
48
- os.makedirs(environ.get('TMP_PATH'), exist_ok=True)
49
- with open(okta_file, 'w') as okta_fh:
50
- okta_fh.write("%s" % token)
51
- return token
52
-
53
-
54
- def get_authentication_token(): # pragma: no cover
55
- okta_file = environ.get('TMP_PATH') + 'okta_token'
56
- token = ''
57
- if path.isfile(okta_file):
58
- one_day_ago = datetime.now() - relativedelta(days=1)
59
- file_time = datetime.fromtimestamp(path.getmtime(okta_file))
60
- if file_time > one_day_ago:
61
- with open(okta_file, 'r') as okta_fh:
62
- token = okta_fh.read().replace("\n", "")
63
- else:
64
- token = update_okta_token()
65
- else:
66
- token = update_okta_token()
67
- return token
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes