wbtools 3.3.0__tar.gz → 3.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wbtools-3.3.0/wbtools.egg-info → wbtools-3.5.0}/PKG-INFO +2 -1
- {wbtools-3.3.0 → wbtools-3.5.0}/pyproject.toml +2 -1
- wbtools-3.5.0/wbtools/literature/abc_search.py +69 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/literature/corpus.py +20 -6
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/literature/paper.py +105 -15
- {wbtools-3.3.0 → wbtools-3.5.0/wbtools.egg-info}/PKG-INFO +2 -1
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools.egg-info/SOURCES.txt +1 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools.egg-info/requires.txt +1 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/LICENSE +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/README.md +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/setup.cfg +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/abstract_manager.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/afp.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/antibody.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/dbmanager.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/expression.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/gene.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/generic.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/paper.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/db/person.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/email/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/email/generic.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/common.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/entity_extraction/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/entity_extraction/email_addresses.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/entity_extraction/ntt_extractor.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/entity_extraction/variations.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/literature_index/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/literature_index/abstract_index.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/literature_index/textpresso.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/text_preprocessing.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/nlp/text_similarity.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/scraping.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/lib/timeout.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/literature/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/literature/person.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/utils/__init__.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/utils/alliance.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools/utils/auth_utils.py +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools.egg-info/dependency_links.txt +0 -0
- {wbtools-3.3.0 → wbtools-3.5.0}/wbtools.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: wbtools
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.5.0
|
|
4
4
|
Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
|
|
5
5
|
Author-email: Valerio Arnaboldi <valearna@caltech.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/WormBase/wbtools
|
|
@@ -25,6 +25,7 @@ Requires-Dist: python-dateutil~=2.8.2
|
|
|
25
25
|
Requires-Dist: python-dotenv~=1.0.0
|
|
26
26
|
Requires-Dist: grobid-client
|
|
27
27
|
Requires-Dist: agr-cognito-py
|
|
28
|
+
Requires-Dist: agr-abc-document-parsers>=1.7.2
|
|
28
29
|
|
|
29
30
|
# WBtools
|
|
30
31
|
> Interface to WormBase curation database and Text Mining functions
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "wbtools"
|
|
7
|
-
version = "3.
|
|
7
|
+
version = "3.5.0"
|
|
8
8
|
authors = [
|
|
9
9
|
{name = "Valerio Arnaboldi", email = "valearna@caltech.edu"},
|
|
10
10
|
]
|
|
@@ -31,6 +31,7 @@ dependencies = [
|
|
|
31
31
|
"python-dotenv~=1.0.0",
|
|
32
32
|
"grobid-client",
|
|
33
33
|
"agr-cognito-py",
|
|
34
|
+
"agr-abc-document-parsers>=1.7.2",
|
|
34
35
|
]
|
|
35
36
|
|
|
36
37
|
[project.urls]
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from typing import Dict, Iterator, List, Tuple
|
|
3
|
+
|
|
4
|
+
import requests
|
|
5
|
+
|
|
6
|
+
from wbtools.literature.paper import ABC_API, ABCRequestError
|
|
7
|
+
from wbtools.utils.auth_utils import get_authentication_token, generate_headers
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
SEARCH_PAGE_SIZE = 100
|
|
12
|
+
WB_PAPER_XREF_PREFIX = "WB:WBPaper"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def get_wb_paper_ids_from_abc(required_workflow_tags: Dict[str, List[str]], date_created_from: str,
|
|
16
|
+
date_created_to: str) -> Iterator[Tuple[str, str]]:
|
|
17
|
+
"""get the WB papers in the ABC corpus created in the given window that have all the required workflow tags,
|
|
18
|
+
newest first
|
|
19
|
+
|
|
20
|
+
Args:
|
|
21
|
+
required_workflow_tags (Dict[str, List[str]]): ABC search facet name (e.g. "file_workflow") -> ATP ids that
|
|
22
|
+
must all be present on the paper
|
|
23
|
+
date_created_from (str): first creation date to include, YYYY-MM-DD
|
|
24
|
+
date_created_to (str): last creation date to include, YYYY-MM-DD
|
|
25
|
+
|
|
26
|
+
Returns:
|
|
27
|
+
Iterator[Tuple[str, str]]: WBPaper id (without the WBPaper prefix) and AGRKB curie of each paper
|
|
28
|
+
|
|
29
|
+
Raises:
|
|
30
|
+
ABCRequestError: if the ABC search fails
|
|
31
|
+
"""
|
|
32
|
+
facets_values = {"mods_in_corpus.keyword": ["WB"]}
|
|
33
|
+
facets_values.update(required_workflow_tags)
|
|
34
|
+
page = 1
|
|
35
|
+
while True:
|
|
36
|
+
hits = _search_references({"facets_values": facets_values,
|
|
37
|
+
"date_created": [date_created_from, date_created_to],
|
|
38
|
+
"sort": [{"date_created": {"order": "desc"}}],
|
|
39
|
+
"size_result_count": SEARCH_PAGE_SIZE, "page": page})
|
|
40
|
+
for hit in hits:
|
|
41
|
+
wb_paper_id = _get_wb_paper_id(hit)
|
|
42
|
+
if wb_paper_id is None:
|
|
43
|
+
logger.warning(f"Skipping ABC reference {hit.get('curie')}: no WBPaper cross-reference")
|
|
44
|
+
continue
|
|
45
|
+
yield wb_paper_id, hit["curie"]
|
|
46
|
+
if len(hits) < SEARCH_PAGE_SIZE:
|
|
47
|
+
return
|
|
48
|
+
page += 1
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _search_references(body: dict) -> list:
|
|
52
|
+
headers = generate_headers(get_authentication_token())
|
|
53
|
+
try:
|
|
54
|
+
response = requests.post(f"https://{ABC_API}/search/references/", json=body, headers=headers, timeout=300)
|
|
55
|
+
response.raise_for_status()
|
|
56
|
+
result = response.json()
|
|
57
|
+
except (requests.exceptions.RequestException, ValueError) as e:
|
|
58
|
+
raise ABCRequestError(f"ABC search failed: {e}") from e
|
|
59
|
+
if not isinstance(result, dict) or result.get("error") or "hits" not in result:
|
|
60
|
+
raise ABCRequestError(f"ABC search failed: {result}")
|
|
61
|
+
return result["hits"]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _get_wb_paper_id(hit: dict):
|
|
65
|
+
for xref in hit.get("cross_references") or []:
|
|
66
|
+
if xref.get("curie", "").startswith(WB_PAPER_XREF_PREFIX) and \
|
|
67
|
+
str(xref.get("is_obsolete")).lower() != "true":
|
|
68
|
+
return xref["curie"][len(WB_PAPER_XREF_PREFIX):]
|
|
69
|
+
return None
|
|
@@ -45,7 +45,8 @@ class CorpusManager(object):
|
|
|
45
45
|
pap_types: List[str] = None,
|
|
46
46
|
exclude_afp_processed: bool = False, exclude_afp_not_curatable: bool = False,
|
|
47
47
|
exclude_no_main_text: bool = False, exclude_no_author_email: bool = False,
|
|
48
|
-
main_file_only: bool = False
|
|
48
|
+
main_file_only: bool = False, text_source: str = "pdf",
|
|
49
|
+
agr_curies: Dict[str, str] = None) -> None:
|
|
49
50
|
"""load papers from WormBase database
|
|
50
51
|
|
|
51
52
|
Args:
|
|
@@ -69,7 +70,14 @@ class CorpusManager(object):
|
|
|
69
70
|
exclude_afp_not_curatable (bool): whether to exclude papers that are not relevant for AFP curation
|
|
70
71
|
exclude_no_main_text (bool): whether to exclude papers without a fulltext that can be converted to txt
|
|
71
72
|
exclude_no_author_email (bool): whether to exclude papers without any contact email in WB
|
|
73
|
+
text_source (str): where to read the text of the papers from when load_pdf_files is True: "pdf" to
|
|
74
|
+
convert the PDF files with GROBID, "abc_markdown" to read the Markdown files
|
|
75
|
+
converted by the ABC
|
|
76
|
+
agr_curies (Dict[str, str]): AGRKB curies of the papers by paper id, e.g. from an ABC search. They take
|
|
77
|
+
precedence over the curies stored in the WB database
|
|
72
78
|
"""
|
|
79
|
+
if text_source not in ("pdf", "abc_markdown"):
|
|
80
|
+
raise ValueError(f"Unknown text_source {text_source}, use 'pdf' or 'abc_markdown'")
|
|
73
81
|
main_db_manager = WBDBManager(db_name, db_user, db_password, db_host)
|
|
74
82
|
with main_db_manager:
|
|
75
83
|
if not paper_ids:
|
|
@@ -96,8 +104,9 @@ class CorpusManager(object):
|
|
|
96
104
|
|
|
97
105
|
for paper_id in paper_ids:
|
|
98
106
|
paper = WBPaper(paper_id=paper_id, db_manager=main_db_manager.paper)
|
|
99
|
-
paper.agr_curie = main_db_manager.paper.get_paper_curie(paper_id)
|
|
107
|
+
paper.agr_curie = (agr_curies or {}).get(paper_id) or main_db_manager.paper.get_paper_curie(paper_id)
|
|
100
108
|
if paper.agr_curie is None:
|
|
109
|
+
logger.warning(f"Skipping paper {paper_id}: no AGRKB curie")
|
|
101
110
|
continue
|
|
102
111
|
if exclude_afp_processed and paper_id in afp_processed_ids:
|
|
103
112
|
logger.info("Skipping paper already processed by AFP")
|
|
@@ -119,7 +128,8 @@ class CorpusManager(object):
|
|
|
119
128
|
continue
|
|
120
129
|
if exclude_no_author_email and not paper.get_authors_with_email_address_in_wb(
|
|
121
130
|
blacklisted_email_addresses=blacklisted_email_addresses):
|
|
122
|
-
logger.info("Skipping paper without any email address in
|
|
131
|
+
logger.info("Skipping paper without any email address in ABC or WB authors with records "
|
|
132
|
+
"in WB")
|
|
123
133
|
continue
|
|
124
134
|
if load_afp_info:
|
|
125
135
|
logger.info("Loading AFP info for paper")
|
|
@@ -127,9 +137,13 @@ class CorpusManager(object):
|
|
|
127
137
|
paper_ids_full_submission=afp_full_submission_ids,
|
|
128
138
|
paper_ids_partial_submission=afp_partial_submission_ids)
|
|
129
139
|
if load_pdf_files:
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
140
|
+
if text_source == "abc_markdown":
|
|
141
|
+
logger.info("Loading text from ABC Markdown files for paper")
|
|
142
|
+
paper.load_text_from_abc_markdown()
|
|
143
|
+
else:
|
|
144
|
+
logger.info("Loading text from PDF files for paper")
|
|
145
|
+
if paper.load_text_from_pdf_files(main_file_only=main_file_only) is False:
|
|
146
|
+
continue
|
|
133
147
|
if exclude_temp_pdf and paper.is_temp():
|
|
134
148
|
logger.info("Skipping proof paper")
|
|
135
149
|
continue
|
|
@@ -17,6 +17,7 @@ from pathlib import Path
|
|
|
17
17
|
from grobid_client.api.pdf import process_fulltext_document
|
|
18
18
|
from grobid_client.models import Article, ProcessForm
|
|
19
19
|
from grobid_client.types import TEI, File
|
|
20
|
+
from agr_abc_document_parsers import extract_sentences, read_markdown
|
|
20
21
|
|
|
21
22
|
|
|
22
23
|
from wbtools.db.afp import WBAFPDBManager
|
|
@@ -69,6 +70,10 @@ def convert_pdf_to_txt(file_path):
|
|
|
69
70
|
return []
|
|
70
71
|
|
|
71
72
|
|
|
73
|
+
class ABCRequestError(Exception):
|
|
74
|
+
"""raised when data that is required to process a paper cannot be retrieved from the ABC"""
|
|
75
|
+
|
|
76
|
+
|
|
72
77
|
def get_data_from_url(url, headers=None, file_type='json'):
|
|
73
78
|
try:
|
|
74
79
|
response = requests.request("GET", url, headers=headers)
|
|
@@ -85,6 +90,22 @@ def get_data_from_url(url, headers=None, file_type='json'):
|
|
|
85
90
|
return None
|
|
86
91
|
|
|
87
92
|
|
|
93
|
+
MARKDOWN_MAIN_FILE_CLASS = "converted_merged_main"
|
|
94
|
+
MARKDOWN_SUPPLEMENT_FILE_CLASS = "converted_merged_supplement"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _decode_markdown(content: bytes) -> str:
|
|
98
|
+
try:
|
|
99
|
+
return content.decode("utf-8")
|
|
100
|
+
except UnicodeDecodeError:
|
|
101
|
+
return content.decode("latin-1", errors="replace")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _is_wb_or_shared_file(ref_file: dict) -> bool:
|
|
105
|
+
mods = ref_file.get("referencefile_mods") or []
|
|
106
|
+
return not mods or any(mod.get("mod_abbreviation") in (None, "WB") for mod in mods)
|
|
107
|
+
|
|
108
|
+
|
|
88
109
|
class WBPaper(object):
|
|
89
110
|
"""WormBase paper information"""
|
|
90
111
|
|
|
@@ -115,6 +136,7 @@ class WBPaper(object):
|
|
|
115
136
|
self.afp_processed = False
|
|
116
137
|
self.afp_partial_submission = False
|
|
117
138
|
self.afp_contact_emails = []
|
|
139
|
+
self.abc_email_addresses = None
|
|
118
140
|
self.db_manager = db_manager
|
|
119
141
|
|
|
120
142
|
def get_corresponding_author(self) -> Union[WBAuthor, None]:
|
|
@@ -213,6 +235,56 @@ class WBPaper(object):
|
|
|
213
235
|
return False
|
|
214
236
|
return added_ref_files > 0
|
|
215
237
|
|
|
238
|
+
def load_text_from_abc_markdown(self) -> bool:
|
|
239
|
+
"""load the main text and the supplements of the paper from the Markdown files converted by the ABC
|
|
240
|
+
|
|
241
|
+
Returns:
|
|
242
|
+
bool: True, the main text has been loaded
|
|
243
|
+
|
|
244
|
+
Raises:
|
|
245
|
+
ABCRequestError: if the list of files, the main Markdown file or its conversion to sentences fails
|
|
246
|
+
"""
|
|
247
|
+
headers = generate_headers(get_authentication_token())
|
|
248
|
+
ref_files = get_data_from_url(f"https://{ABC_API}/reference/referencefile/show_all/{self.agr_curie}", headers)
|
|
249
|
+
if ref_files is None:
|
|
250
|
+
raise ABCRequestError(f"Could not list the ABC files of paper {self.paper_id} ({self.agr_curie})")
|
|
251
|
+
markdown_files = [ref_file for ref_file in ref_files if ref_file.get("file_extension") == "md"]
|
|
252
|
+
main_files = [ref_file for ref_file in markdown_files
|
|
253
|
+
if ref_file.get("file_class") == MARKDOWN_MAIN_FILE_CLASS]
|
|
254
|
+
if not main_files:
|
|
255
|
+
raise ABCRequestError(f"No converted Markdown file in ABC for paper {self.paper_id} ({self.agr_curie})")
|
|
256
|
+
main_file = next((ref_file for ref_file in main_files if _is_wb_or_shared_file(ref_file)), main_files[0])
|
|
257
|
+
try:
|
|
258
|
+
main_text = self._get_sentences_from_abc_markdown(main_file, headers)
|
|
259
|
+
except Exception as e:
|
|
260
|
+
raise ABCRequestError(f"Could not load the converted Markdown file of paper {self.paper_id} "
|
|
261
|
+
f"({self.agr_curie}): {e}") from e
|
|
262
|
+
if not main_text:
|
|
263
|
+
raise ABCRequestError(f"The converted Markdown file of paper {self.paper_id} ({self.agr_curie}) "
|
|
264
|
+
f"has no text")
|
|
265
|
+
self.main_text = main_text
|
|
266
|
+
for supplement in markdown_files:
|
|
267
|
+
if supplement.get("file_class") != MARKDOWN_SUPPLEMENT_FILE_CLASS or \
|
|
268
|
+
not _is_wb_or_shared_file(supplement):
|
|
269
|
+
continue
|
|
270
|
+
try:
|
|
271
|
+
supplement_text = self._get_sentences_from_abc_markdown(supplement, headers)
|
|
272
|
+
except Exception as e:
|
|
273
|
+
logger.warning(f"Skipping supplement {supplement.get('display_name')} of paper {self.paper_id}: {e}")
|
|
274
|
+
continue
|
|
275
|
+
if supplement_text:
|
|
276
|
+
self.supplemental_docs.append(supplement_text)
|
|
277
|
+
return True
|
|
278
|
+
|
|
279
|
+
@staticmethod
|
|
280
|
+
def _get_sentences_from_abc_markdown(ref_file: dict, headers: dict) -> List[str]:
|
|
281
|
+
# file_type='pdf' makes get_data_from_url return the raw bytes of the file
|
|
282
|
+
content = get_data_from_url(f"https://{ABC_API}/reference/referencefile/download_file/"
|
|
283
|
+
f"{ref_file['referencefile_id']}", headers, file_type='pdf')
|
|
284
|
+
if not content:
|
|
285
|
+
raise ValueError(f"download of ABC file {ref_file['referencefile_id']} failed")
|
|
286
|
+
return extract_sentences(read_markdown(_decode_markdown(content)))
|
|
287
|
+
|
|
216
288
|
def load_curation_info_from_db(self):
|
|
217
289
|
"""load curation data from WormBase database"""
|
|
218
290
|
if self.db_manager:
|
|
@@ -322,46 +394,64 @@ class WBPaper(object):
|
|
|
322
394
|
def extract_all_email_addresses_from_text_and_write_to_db(self):
|
|
323
395
|
self.write_email_addresses_to_db(self.extract_all_email_addresses_from_text())
|
|
324
396
|
|
|
397
|
+
def get_abc_email_addresses(self) -> Union[List[str], None]:
|
|
398
|
+
"""get the email addresses associated with the paper in the ABC
|
|
399
|
+
|
|
400
|
+
Returns:
|
|
401
|
+
Union[List[str], None]: the list of email addresses, or None if the ABC request failed
|
|
402
|
+
"""
|
|
403
|
+
if self.abc_email_addresses is None:
|
|
404
|
+
reference_emails_api = f"https://{ABC_API}/reference/{self.agr_curie}/emails"
|
|
405
|
+
headers = generate_headers(get_authentication_token())
|
|
406
|
+
reference_emails = get_data_from_url(reference_emails_api, headers)
|
|
407
|
+
if reference_emails is not None:
|
|
408
|
+
self.abc_email_addresses = [reference_email["email_address"] for reference_email in
|
|
409
|
+
reference_emails]
|
|
410
|
+
return self.abc_email_addresses
|
|
411
|
+
|
|
325
412
|
def get_aut_class_value_for_datatype(self, datatype: str):
|
|
326
413
|
return self.aut_class_values[datatype] if self.aut_class_values[datatype] else None
|
|
327
414
|
|
|
328
415
|
def get_authors_with_email_address_in_wb(self, blacklisted_email_addresses: List[str] = None,
|
|
329
416
|
first_only: bool = False) -> Union[List[Tuple[WBAuthor, str]], None]:
|
|
330
417
|
"""
|
|
331
|
-
Get the
|
|
332
|
-
and the email address
|
|
418
|
+
Get the email addresses associated with the paper in the ABC that have a corresponding person entry in WB and
|
|
419
|
+
return the person object and the email address from the ABC, which may be more recent than the one in WB
|
|
333
420
|
|
|
334
421
|
Args:
|
|
335
422
|
blacklisted_email_addresses (List[str]): a list of email addresses to be excluded from the search
|
|
336
423
|
first_only (bool): whether to return only the first available author
|
|
337
424
|
|
|
338
425
|
Returns:
|
|
339
|
-
Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address
|
|
426
|
+
Union[List[Tuple[WBPerson, str]], None]: a tuple containing the WBPerson and the email address from the ABC.
|
|
340
427
|
If no email is found with a corresponding person in WB, then the function
|
|
341
428
|
will return the corresponding author associated with the paper in WB and
|
|
342
429
|
its email address, if any, otherwise None.
|
|
430
|
+
|
|
431
|
+
Raises:
|
|
432
|
+
ABCRequestError: if the email addresses cannot be retrieved from the ABC
|
|
343
433
|
"""
|
|
344
434
|
result = []
|
|
345
|
-
|
|
346
|
-
if
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
blacklisted_email_addresses = set(blacklisted_email_addresses)
|
|
350
|
-
if
|
|
351
|
-
for
|
|
352
|
-
if "'" not in
|
|
435
|
+
abc_addresses = self.get_abc_email_addresses()
|
|
436
|
+
if abc_addresses is None:
|
|
437
|
+
raise ABCRequestError(f"Could not retrieve email addresses from ABC for paper {self.paper_id} "
|
|
438
|
+
f"({self.agr_curie})")
|
|
439
|
+
blacklisted_email_addresses = set(blacklisted_email_addresses or [])
|
|
440
|
+
if abc_addresses:
|
|
441
|
+
for abc_address in abc_addresses:
|
|
442
|
+
if "'" not in abc_address:
|
|
353
443
|
person_id = self.db_manager.get_db_manager(
|
|
354
|
-
WBPersonDBManager).get_person_id_from_email_address(
|
|
444
|
+
WBPersonDBManager).get_person_id_from_email_address(abc_address)
|
|
355
445
|
if person_id:
|
|
356
446
|
current_address = self.db_manager.get_db_manager(
|
|
357
447
|
WBPersonDBManager).get_current_email_address_for_person(person_id)
|
|
358
448
|
if current_address and current_address not in blacklisted_email_addresses:
|
|
359
449
|
result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
|
|
360
450
|
person_id=person_id), current_address))
|
|
361
|
-
if (
|
|
362
|
-
|
|
451
|
+
if (abc_address not in blacklisted_email_addresses and
|
|
452
|
+
abc_address != current_address):
|
|
363
453
|
result.append((self.db_manager.get_db_manager(WBPersonDBManager).get_person(
|
|
364
|
-
person_id=person_id),
|
|
454
|
+
person_id=person_id), abc_address))
|
|
365
455
|
if first_only and result:
|
|
366
456
|
return result
|
|
367
457
|
if not result:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: wbtools
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.5.0
|
|
4
4
|
Summary: Interface to WormBase (www.wormbase.org) curation data, including literature management and NLP functions
|
|
5
5
|
Author-email: Valerio Arnaboldi <valearna@caltech.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/WormBase/wbtools
|
|
@@ -25,6 +25,7 @@ Requires-Dist: python-dateutil~=2.8.2
|
|
|
25
25
|
Requires-Dist: python-dotenv~=1.0.0
|
|
26
26
|
Requires-Dist: grobid-client
|
|
27
27
|
Requires-Dist: agr-cognito-py
|
|
28
|
+
Requires-Dist: agr-abc-document-parsers>=1.7.2
|
|
28
29
|
|
|
29
30
|
# WBtools
|
|
30
31
|
> Interface to WormBase curation database and Text Mining functions
|
|
@@ -34,6 +34,7 @@ wbtools/lib/nlp/literature_index/__init__.py
|
|
|
34
34
|
wbtools/lib/nlp/literature_index/abstract_index.py
|
|
35
35
|
wbtools/lib/nlp/literature_index/textpresso.py
|
|
36
36
|
wbtools/literature/__init__.py
|
|
37
|
+
wbtools/literature/abc_search.py
|
|
37
38
|
wbtools/literature/corpus.py
|
|
38
39
|
wbtools/literature/paper.py
|
|
39
40
|
wbtools/literature/person.py
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|