wbtools 3.2.0__tar.gz → 3.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wbtools-3.2.0/wbtools.egg-info → wbtools-3.4.0}/PKG-INFO +2 -1
- {wbtools-3.2.0 → wbtools-3.4.0}/pyproject.toml +2 -1
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/corpus.py +12 -5
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/paper.py +43 -17
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/utils/alliance.py +1 -1
- wbtools-3.4.0/wbtools/utils/auth_utils.py +21 -0
- {wbtools-3.2.0 → wbtools-3.4.0/wbtools.egg-info}/PKG-INFO +2 -1
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/SOURCES.txt +1 -1
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/requires.txt +1 -0
- wbtools-3.2.0/wbtools/utils/okta_utils.py +0 -67
- {wbtools-3.2.0 → wbtools-3.4.0}/LICENSE +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/README.md +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/setup.cfg +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/abstract_manager.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/afp.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/antibody.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/dbmanager.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/expression.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/gene.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/generic.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/paper.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/db/person.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/email/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/email/generic.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/common.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/email_addresses.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/ntt_extractor.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/entity_extraction/variations.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/abstract_index.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/literature_index/textpresso.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_preprocessing.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/nlp/text_similarity.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/scraping.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/lib/timeout.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/literature/person.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools/utils/__init__.py +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/dependency_links.txt +0 -0
- {wbtools-3.2.0 → wbtools-3.4.0}/wbtools.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: wbtools
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.4.0
|
|
4
4
|
Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
|
|
5
5
|
Author-email: Valerio Arnaboldi <valearna@caltech.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/WormBase/wbtools
|
|
@@ -24,6 +24,7 @@ Requires-Dist: requests~=2.31.0
|
|
|
24
24
|
Requires-Dist: python-dateutil~=2.8.2
|
|
25
25
|
Requires-Dist: python-dotenv~=1.0.0
|
|
26
26
|
Requires-Dist: grobid-client
|
|
27
|
+
Requires-Dist: agr-cognito-py
|
|
27
28
|
|
|
28
29
|
# WBtools
|
|
29
30
|
> Interface to WormBase curation database and Text Mining functions
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "wbtools"
|
|
7
|
-
version = "3.
|
|
7
|
+
version = "3.4.0"
|
|
8
8
|
authors = [
|
|
9
9
|
{name = "Valerio Arnaboldi", email = "valearna@caltech.edu"},
|
|
10
10
|
]
|
|
@@ -30,6 +30,7 @@ dependencies = [
|
|
|
30
30
|
"python-dateutil~=2.8.2",
|
|
31
31
|
"python-dotenv~=1.0.0",
|
|
32
32
|
"grobid-client",
|
|
33
|
+
"agr-cognito-py",
|
|
33
34
|
]
|
|
34
35
|
|
|
35
36
|
[project.urls]
|
|
@@ -8,7 +8,7 @@ from wbtools.db.dbmanager import WBDBManager
|
|
|
8
8
|
from wbtools.lib.nlp.common import PaperSections
|
|
9
9
|
from wbtools.lib.nlp.text_preprocessing import preprocess
|
|
10
10
|
from wbtools.lib.nlp.text_similarity import get_softcosine_index, get_similar_documents, SimilarityResult
|
|
11
|
-
from wbtools.literature.paper import WBPaper
|
|
11
|
+
from wbtools.literature.paper import WBPaper, ABCRequestError
|
|
12
12
|
|
|
13
13
|
|
|
14
14
|
logger = logging.getLogger(__name__)
|
|
@@ -117,10 +117,17 @@ class CorpusManager(object):
|
|
|
117
117
|
logger.info("Loading bib info for paper")
|
|
118
118
|
if paper.load_bib_info() is False:
|
|
119
119
|
continue
|
|
120
|
-
if exclude_no_author_email
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
120
|
+
if exclude_no_author_email:
|
|
121
|
+
try:
|
|
122
|
+
authors_with_email = paper.get_authors_with_email_address_in_wb(
|
|
123
|
+
blacklisted_email_addresses=blacklisted_email_addresses)
|
|
124
|
+
except ABCRequestError as e:
|
|
125
|
+
logger.warning(f"Skipping paper: {e}")
|
|
126
|
+
continue
|
|
127
|
+
if not authors_with_email:
|
|
128
|
+
logger.info("Skipping paper without any email address in ABC or WB authors with records "
|
|
129
|
+
"in WB")
|
|
130
|
+
continue
|
|
124
131
|
if load_afp_info:
|
|
125
132
|
logger.info("Loading AFP info for paper")
|
|
126
133
|
paper.load_afp_info_from_db(paper_ids_no_submission=afp_no_submission_ids,
|
|
@@ -26,7 +26,7 @@ from wbtools.lib.nlp.entity_extraction.email_addresses import get_email_addresse
|
|
|
26
26
|
from wbtools.lib.nlp.text_preprocessing import preprocess, get_documents_from_text, PaperSections
|
|
27
27
|
from wbtools.lib.timeout import timeout
|
|
28
28
|
from wbtools.literature.person import WBAuthor
|
|
29
|
-
from wbtools.utils.
|
|
29
|
+
from wbtools.utils.auth_utils import get_authentication_token, generate_headers
|
|
30
30
|
|
|
31
31
|
logger = logging.getLogger(__name__)
|
|
32
32
|
|
|
@@ -69,6 +69,10 @@ def convert_pdf_to_txt(file_path):
|
|
|
69
69
|
return []
|
|
70
70
|
|
|
71
71
|
|
|
72
|
+
class ABCRequestError(Exception):
|
|
73
|
+
"""raised when data that is required to process a paper cannot be retrieved from the ABC"""
|
|
74
|
+
|
|
75
|
+
|
|
72
76
|
def get_data_from_url(url, headers=None, file_type='json'):
|
|
73
77
|
try:
|
|
74
78
|
response = requests.request("GET", url, headers=headers)
|
|
@@ -115,6 +119,7 @@ class WBPaper(object):
|
|
|
115
119
|
self.afp_processed = False
|
|
116
120
|
self.afp_partial_submission = False
|
|
117
121
|
self.afp_contact_emails = []
|
|
122
|
+
self.abc_email_addresses = None
|
|
118
123
|
self.db_manager = db_manager
|
|
119
124
|
|
|
120
125
|
def get_corresponding_author(self) -> Union[WBAuthor, None]:
|
|
@@ -193,6 +198,7 @@ class WBPaper(object):
|
|
|
193
198
|
blue_api_base_url = os.environ.get('API_SERVER', "literature-rest.alliancegenome.org")
|
|
194
199
|
all_reffiles_for_pap_api = f'https://{blue_api_base_url}/reference/referencefile/show_all/{self.agr_curie}'
|
|
195
200
|
request = urllib.request.Request(url=all_reffiles_for_pap_api)
|
|
201
|
+
request.add_header("Authorization", f"Bearer {get_authentication_token()}")
|
|
196
202
|
request.add_header("Content-type", "application/json")
|
|
197
203
|
request.add_header("Accept", "application/json")
|
|
198
204
|
added_ref_files = 0
|
|
@@ -227,7 +233,9 @@ class WBPaper(object):
|
|
|
227
233
|
raise Exception("PaperDBManager not set")
|
|
228
234
|
# Get Alliance reference info from WBPaperID
|
|
229
235
|
ref_info_from_xref_api = f"https://{ABC_API}/reference/by_cross_reference/WB:WBPaper{self.paper_id}"
|
|
230
|
-
|
|
236
|
+
token = get_authentication_token()
|
|
237
|
+
headers = generate_headers(token)
|
|
238
|
+
ref_info: dict = get_data_from_url(ref_info_from_xref_api, headers)
|
|
231
239
|
if ref_info:
|
|
232
240
|
self.abstract = ref_info["abstract"]
|
|
233
241
|
self.title = ref_info["title"]
|
|
@@ -319,46 +327,64 @@ class WBPaper(object):
|
|
|
319
327
|
def extract_all_email_addresses_from_text_and_write_to_db(self):
|
|
320
328
|
self.write_email_addresses_to_db(self.extract_all_email_addresses_from_text())
|
|
321
329
|
|
|
330
|
+
def get_abc_email_addresses(self) -> Union[List[str], None]:
|
|
331
|
+
"""get the email addresses associated with the paper in the ABC
|
|
332
|
+
|
|
333
|
+
Returns:
|
|
334
|
+
Union[List[str], None]: the list of email addresses, or None if the ABC request failed
|
|
335
|
+
"""
|
|
336
|
+
if self.abc_email_addresses is None:
|
|
337
|
+
reference_emails_api = f"https://{ABC_API}/reference/{self.agr_curie}/emails"
|
|
338
|
+
headers = generate_headers(get_authentication_token())
|
|
339
|
+
reference_emails = get_data_from_url(reference_emails_api, headers)
|
|
340
|
+
if reference_emails is not None:
|
|
341
|
+
self.abc_email_addresses = [reference_email["email_address"] for reference_email in
|
|
342
|
+
reference_emails]
|
|
343
|
+
return self.abc_email_addresses
|
|
344
|
+
|
|
322
345
|
def get_aut_class_value_for_datatype(self, datatype: str):
|
|
323
346
|
return self.aut_class_values[datatype] if self.aut_class_values[datatype] else None
|
|
324
347
|
|
|
325
348
|
def get_authors_with_email_address_in_wb(self, blacklisted_email_addresses: List[str] = None,
|
|
326
349
|
first_only: bool = False) -> Union[List[Tuple[WBAuthor, str]], None]:
|
|
327
350
|
"""
|
|
328
|
-
Get the
|
|
329
|
-
and the email address
|
|
351
|
+
Get the email addresses associated with the paper in the ABC that have a corresponding person entry in WB and
|
|
352
|
+
return the person object and the email address from the ABC, which may be more recent than the one in WB
|
|
330
353
|
|
|
331
354
|
Args:
|
|
332
355
|
blacklisted_email_addresses (List[str]): a list of email addresses to be excluded from the search
|
|
333
356
|
first_only (bool): whether to return only the first available author
|
|
334
357
|
|
|
335
358
|
Returns:
|
|
336
|
-
Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address
|
|
359
|
+
Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address from the ABC.
|
|
337
360
|
If no email is found with a corresponding person in WB, then the function
|
|
338
361
|
will return the corresponding author associated with the paper in WB and
|
|
339
362
|
its email address, if any, otherwise None.
|
|
363
|
+
|
|
364
|
+
Raises:
|
|
365
|
+
ABCRequestError: if the email addresses cannot be retrieved from the ABC
|
|
340
366
|
"""
|
|
341
367
|
result = []
|
|
342
|
-
|
|
343
|
-
if
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
blacklisted_email_addresses = set(blacklisted_email_addresses)
|
|
347
|
-
if
|
|
348
|
-
for
|
|
349
|
-
if "'" not in
|
|
368
|
+
abc_addresses = self.get_abc_email_addresses()
|
|
369
|
+
if abc_addresses is None:
|
|
370
|
+
raise ABCRequestError(f"Could not retrieve email addresses from ABC for paper {self.paper_id} "
|
|
371
|
+
f"({self.agr_curie})")
|
|
372
|
+
blacklisted_email_addresses = set(blacklisted_email_addresses or [])
|
|
373
|
+
if abc_addresses:
|
|
374
|
+
for abc_address in abc_addresses:
|
|
375
|
+
if "'" not in abc_address:
|
|
350
376
|
person_id = self.db_manager.get_db_manager(
|
|
351
|
-
WBPersonDBManager).get_person_id_from_email_address(
|
|
377
|
+
WBPersonDBManager).get_person_id_from_email_address(abc_address)
|
|
352
378
|
if person_id:
|
|
353
379
|
current_address = self.db_manager.get_db_manager(
|
|
354
380
|
WBPersonDBManager).get_current_email_address_for_person(person_id)
|
|
355
381
|
if current_address and current_address not in blacklisted_email_addresses:
|
|
356
382
|
result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
|
|
357
383
|
person_id=person_id), current_address))
|
|
358
|
-
if (
|
|
359
|
-
|
|
384
|
+
if (abc_address not in blacklisted_email_addresses and
|
|
385
|
+
abc_address != current_address):
|
|
360
386
|
result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
|
|
361
|
-
person_id=person_id),
|
|
387
|
+
person_id=person_id), abc_address))
|
|
362
388
|
if first_only and result:
|
|
363
389
|
return result
|
|
364
390
|
if not result:
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Authentication utilities for AGR API access.
|
|
3
|
+
|
|
4
|
+
This module provides authentication token generation using AWS Cognito
|
|
5
|
+
via the agr_cognito_py library. The functions are re-exported for
|
|
6
|
+
backward compatibility with code that was using the old Okta-based
|
|
7
|
+
authentication.
|
|
8
|
+
|
|
9
|
+
Required environment variables for Cognito authentication:
|
|
10
|
+
- COGNITO_ADMIN_CLIENT_ID
|
|
11
|
+
- COGNITO_ADMIN_CLIENT_SECRET
|
|
12
|
+
- COGNITO_TOKEN_URL
|
|
13
|
+
- COGNITO_ADMIN_SCOPE
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from agr_cognito_py import (
|
|
17
|
+
get_admin_token as get_authentication_token,
|
|
18
|
+
generate_headers
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
__all__ = ['get_authentication_token', 'generate_headers']
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: wbtools
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.4.0
|
|
4
4
|
Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
|
|
5
5
|
Author-email: Valerio Arnaboldi <valearna@caltech.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/WormBase/wbtools
|
|
@@ -24,6 +24,7 @@ Requires-Dist: requests~=2.31.0
|
|
|
24
24
|
Requires-Dist: python-dateutil~=2.8.2
|
|
25
25
|
Requires-Dist: python-dotenv~=1.0.0
|
|
26
26
|
Requires-Dist: grobid-client
|
|
27
|
+
Requires-Dist: agr-cognito-py
|
|
27
28
|
|
|
28
29
|
# WBtools
|
|
29
30
|
> Interface to WormBase curation database and Text Mining functions
|
|
@@ -1,67 +0,0 @@
|
|
|
1
|
-
import os
|
|
2
|
-
from json import loads
|
|
3
|
-
from os import environ, path
|
|
4
|
-
import logging
|
|
5
|
-
import logging.config
|
|
6
|
-
from datetime import datetime
|
|
7
|
-
from dateutil.relativedelta import relativedelta
|
|
8
|
-
import requests
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
def generate_headers(token):
|
|
12
|
-
"""
|
|
13
|
-
|
|
14
|
-
:param token:
|
|
15
|
-
:return:
|
|
16
|
-
"""
|
|
17
|
-
|
|
18
|
-
authorization = 'Bearer ' + token
|
|
19
|
-
headers = {
|
|
20
|
-
'Authorization': authorization,
|
|
21
|
-
'Content-Type': 'application/json',
|
|
22
|
-
'Accept': 'application/json'
|
|
23
|
-
}
|
|
24
|
-
return headers
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
def update_okta_token(): # pragma: no cover
|
|
28
|
-
"""
|
|
29
|
-
|
|
30
|
-
:return:
|
|
31
|
-
"""
|
|
32
|
-
|
|
33
|
-
url = f"https://{os.environ.get('OKTA_DOMAIN')}/v1/token"
|
|
34
|
-
headers = {
|
|
35
|
-
'Content-Type': 'application/x-www-form-urlencoded',
|
|
36
|
-
'Accept': 'application/json'
|
|
37
|
-
}
|
|
38
|
-
data_dict = dict()
|
|
39
|
-
data_dict['grant_type'] = 'client_credentials'
|
|
40
|
-
data_dict['client_id'] = environ.get('OKTA_CLIENT_ID')
|
|
41
|
-
data_dict['client_secret'] = environ.get('OKTA_CLIENT_SECRET')
|
|
42
|
-
data_dict['scope'] = 'admin'
|
|
43
|
-
post_return = requests.post(url, headers=headers, data=data_dict)
|
|
44
|
-
logging.warning(post_return.text)
|
|
45
|
-
response_dict = loads(post_return.text)
|
|
46
|
-
token = response_dict['access_token']
|
|
47
|
-
okta_file = environ.get('TMP_PATH') + 'okta_token'
|
|
48
|
-
os.makedirs(environ.get('TMP_PATH'), exist_ok=True)
|
|
49
|
-
with open(okta_file, 'w') as okta_fh:
|
|
50
|
-
okta_fh.write("%s" % token)
|
|
51
|
-
return token
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
def get_authentication_token(): # pragma: no cover
|
|
55
|
-
okta_file = environ.get('TMP_PATH') + 'okta_token'
|
|
56
|
-
token = ''
|
|
57
|
-
if path.isfile(okta_file):
|
|
58
|
-
one_day_ago = datetime.now() - relativedelta(days=1)
|
|
59
|
-
file_time = datetime.fromtimestamp(path.getmtime(okta_file))
|
|
60
|
-
if file_time > one_day_ago:
|
|
61
|
-
with open(okta_file, 'r') as okta_fh:
|
|
62
|
-
token = okta_fh.read().replace("\n", "")
|
|
63
|
-
else:
|
|
64
|
-
token = update_okta_token()
|
|
65
|
-
else:
|
|
66
|
-
token = update_okta_token()
|
|
67
|
-
return token
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|