scholarlib 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scholarlib/__init__.py +3 -0
- scholarlib/adapters.py +94 -0
- scholarlib/apis/__init__.py +0 -0
- scholarlib/apis/core_api.py +41 -0
- scholarlib/apis/crossref.py +93 -0
- scholarlib/apis/openaire.py +88 -0
- scholarlib/apis/openalex.py +199 -0
- scholarlib/apis/orcid.py +118 -0
- scholarlib/apis/semantic_scholar.py +99 -0
- scholarlib/apis/unpaywall.py +47 -0
- scholarlib/apis/zotero_local.py +156 -0
- scholarlib/cli/__init__.py +0 -0
- scholarlib/cli/jstor_index.py +165 -0
- scholarlib/cli/litreview.py +326 -0
- scholarlib/cli/scholarfocus.py +662 -0
- scholarlib/config.example.yaml +68 -0
- scholarlib/config.py +172 -0
- scholarlib/dedup.py +177 -0
- scholarlib/http/__init__.py +0 -0
- scholarlib/http/base.py +212 -0
- scholarlib/http/budget.py +162 -0
- scholarlib/http/cache.py +144 -0
- scholarlib/http/policy.py +59 -0
- scholarlib/jstor/__init__.py +0 -0
- scholarlib/jstor/build.py +336 -0
- scholarlib/jstor/indexes.sql +8 -0
- scholarlib/jstor/query.py +295 -0
- scholarlib/jstor/schema.sql +48 -0
- scholarlib/pipeline/__init__.py +0 -0
- scholarlib/pipeline/abstracts.py +59 -0
- scholarlib/pipeline/cluster.py +117 -0
- scholarlib/pipeline/context.py +92 -0
- scholarlib/pipeline/disambiguate.py +377 -0
- scholarlib/pipeline/jstor_coverage.py +137 -0
- scholarlib/pipeline/oa_links.py +61 -0
- scholarlib/pipeline/profile.py +333 -0
- scholarlib/pipeline/rank.py +139 -0
- scholarlib/pipeline/seed.py +97 -0
- scholarlib/pipeline/snowball.py +120 -0
- scholarlib/progress.py +131 -0
- scholarlib/records.py +202 -0
- scholarlib/render/__init__.py +0 -0
- scholarlib/render/bib.py +139 -0
- scholarlib/render/json_out.py +120 -0
- scholarlib/render/narrative.py +167 -0
- scholarlib/render/systematic.py +93 -0
- scholarlib-0.3.0.dist-info/METADATA +232 -0
- scholarlib-0.3.0.dist-info/RECORD +52 -0
- scholarlib-0.3.0.dist-info/WHEEL +5 -0
- scholarlib-0.3.0.dist-info/entry_points.txt +4 -0
- scholarlib-0.3.0.dist-info/licenses/LICENSE +21 -0
- scholarlib-0.3.0.dist-info/top_level.txt +1 -0
scholarlib/__init__.py
ADDED
scholarlib/adapters.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Convert each source's native shape into the canonical Record."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from scholarlib.apis.openalex import reconstruct_abstract
|
|
8
|
+
from scholarlib.records import Record
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _authors_from_openalex(work: dict) -> list[str]:
|
|
12
|
+
out = []
|
|
13
|
+
for a in work.get("authorships") or []:
|
|
14
|
+
name = ((a.get("author") or {}).get("display_name"))
|
|
15
|
+
if name:
|
|
16
|
+
out.append(name)
|
|
17
|
+
return out
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _topics_from_openalex(work: dict) -> list[str]:
|
|
21
|
+
names = []
|
|
22
|
+
for t in (work.get("topics") or [])[:5]:
|
|
23
|
+
if t.get("display_name"):
|
|
24
|
+
names.append(t["display_name"])
|
|
25
|
+
for c in (work.get("concepts") or [])[:5]:
|
|
26
|
+
if c.get("display_name") and c["display_name"] not in names:
|
|
27
|
+
names.append(c["display_name"])
|
|
28
|
+
return names
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def from_openalex_work(work: dict, *, provenance: Optional[str] = None) -> Record:
|
|
32
|
+
abstract = reconstruct_abstract(work.get("abstract_inverted_index"))
|
|
33
|
+
loc = work.get("primary_location") or {}
|
|
34
|
+
src = loc.get("source") or {}
|
|
35
|
+
best_oa = work.get("best_oa_location") or {}
|
|
36
|
+
oa = work.get("open_access") or {}
|
|
37
|
+
return Record(
|
|
38
|
+
doi=work.get("doi"),
|
|
39
|
+
openalex_id=work.get("id"),
|
|
40
|
+
title=work.get("title") or work.get("display_name"),
|
|
41
|
+
authors=_authors_from_openalex(work),
|
|
42
|
+
year=work.get("publication_year"),
|
|
43
|
+
venue=src.get("display_name"),
|
|
44
|
+
type=work.get("type"),
|
|
45
|
+
language=work.get("language"),
|
|
46
|
+
abstract=abstract,
|
|
47
|
+
abstract_source="openalex" if abstract else None,
|
|
48
|
+
cited_by_count=work.get("cited_by_count"),
|
|
49
|
+
referenced_works=list(work.get("referenced_works") or []),
|
|
50
|
+
related_works=list(work.get("related_works") or []),
|
|
51
|
+
topics=_topics_from_openalex(work),
|
|
52
|
+
oa_status=oa.get("oa_status"),
|
|
53
|
+
oa_url=best_oa.get("pdf_url") or best_oa.get("landing_page_url"),
|
|
54
|
+
provenance=[provenance] if provenance else [],
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def from_crossref_item(item: dict, *, provenance: str = "crossref") -> Record:
|
|
59
|
+
titles = item.get("title") or []
|
|
60
|
+
containers = item.get("container-title") or []
|
|
61
|
+
authors = []
|
|
62
|
+
for a in item.get("author") or []:
|
|
63
|
+
nm = " ".join(x for x in (a.get("given"), a.get("family")) if x) or a.get("name")
|
|
64
|
+
if nm:
|
|
65
|
+
authors.append(nm)
|
|
66
|
+
issued = ((item.get("issued") or {}).get("date-parts") or [[None]])[0]
|
|
67
|
+
return Record(
|
|
68
|
+
doi=item.get("DOI"),
|
|
69
|
+
title=titles[0] if titles else None,
|
|
70
|
+
authors=authors,
|
|
71
|
+
year=issued[0] if issued else None,
|
|
72
|
+
venue=containers[0] if containers else None,
|
|
73
|
+
type=item.get("type"),
|
|
74
|
+
cited_by_count=item.get("is-referenced-by-count"),
|
|
75
|
+
provenance=[provenance],
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def from_s2_paper(paper: dict, *, provenance: str = "semantic_scholar") -> Record:
|
|
80
|
+
ext = paper.get("externalIds") or {}
|
|
81
|
+
tldr = (paper.get("tldr") or {}).get("text")
|
|
82
|
+
abstract = paper.get("abstract") or tldr
|
|
83
|
+
return Record(
|
|
84
|
+
doi=ext.get("DOI"),
|
|
85
|
+
title=paper.get("title"),
|
|
86
|
+
year=paper.get("year"),
|
|
87
|
+
venue=paper.get("venue"),
|
|
88
|
+
abstract=abstract,
|
|
89
|
+
abstract_source=("semantic_scholar" if paper.get("abstract")
|
|
90
|
+
else ("s2_tldr" if tldr else None)),
|
|
91
|
+
cited_by_count=paper.get("citationCount"),
|
|
92
|
+
topics=list(paper.get("fieldsOfStudy") or []),
|
|
93
|
+
provenance=[provenance],
|
|
94
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""CORE client. Key is optional but raises the rate limit from 10/min to 150/min."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
from scholarlib.http.base import BaseClient
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
BASE_URL = "https://api.core.ac.uk/v3"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class COREClient(BaseClient):
|
|
16
|
+
name = "core"
|
|
17
|
+
base_url = BASE_URL
|
|
18
|
+
|
|
19
|
+
def __init__(self, api_key: Optional[str] = None, **kw):
|
|
20
|
+
super().__init__(**kw)
|
|
21
|
+
self.api_key = api_key
|
|
22
|
+
if not api_key:
|
|
23
|
+
logger.debug("CORE: no API key — rate limit is 10/min instead of 150/min")
|
|
24
|
+
|
|
25
|
+
def _auth_headers(self) -> dict:
|
|
26
|
+
return {"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}
|
|
27
|
+
|
|
28
|
+
def search_works(self, query: str, limit: int = 10) -> list[dict]:
|
|
29
|
+
data = self.get("/search/works", {"q": query, "limit": limit})
|
|
30
|
+
return (data or {}).get("results", [])
|
|
31
|
+
|
|
32
|
+
def get_work_by_doi(self, doi: str) -> Optional[dict]:
|
|
33
|
+
d = str(doi).replace("https://doi.org/", "").strip()
|
|
34
|
+
results = self.search_works(f'doi:"{d}"', limit=1)
|
|
35
|
+
return results[0] if results else None
|
|
36
|
+
|
|
37
|
+
def get_abstract(self, doi: str) -> Optional[str]:
|
|
38
|
+
w = self.get_work_by_doi(doi)
|
|
39
|
+
if not w:
|
|
40
|
+
return None
|
|
41
|
+
return w.get("abstract") or w.get("description") or None
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""CrossRef client. Free, polite pool via a mailto in the User-Agent."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
from scholarlib.http.base import BaseClient
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
BASE_URL = "https://api.crossref.org"
|
|
13
|
+
|
|
14
|
+
# Verified against the live API. `reference-count` is NOT a valid select field
|
|
15
|
+
# and returns HTTP 400 select-not-available -- that was the long-standing bug.
|
|
16
|
+
# The valid spellings are `references-count` and `is-referenced-by-count`.
|
|
17
|
+
SAFE_SELECT = (
|
|
18
|
+
"DOI,title,author,issued,type,container-title,"
|
|
19
|
+
"references-count,is-referenced-by-count,abstract"
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class CrossRefClient(BaseClient):
|
|
24
|
+
name = "crossref"
|
|
25
|
+
base_url = BASE_URL
|
|
26
|
+
|
|
27
|
+
def __init__(self, email: Optional[str] = None, **kw):
|
|
28
|
+
super().__init__(contact_email=email, **kw)
|
|
29
|
+
self.email = email
|
|
30
|
+
|
|
31
|
+
def _auth_params(self) -> dict:
|
|
32
|
+
return {"mailto": self.email} if self.email else {}
|
|
33
|
+
|
|
34
|
+
def get_work_by_doi(self, doi: str) -> Optional[dict]:
|
|
35
|
+
d = str(doi).replace("https://doi.org/", "").strip()
|
|
36
|
+
data = self.get(f"/works/{d}")
|
|
37
|
+
return (data or {}).get("message")
|
|
38
|
+
|
|
39
|
+
def search_works_by_author(self, name: str, rows: int = 50,
|
|
40
|
+
*, select: Optional[str] = None) -> list[dict]:
|
|
41
|
+
params: dict = {"query.author": name, "rows": min(rows, 1000)}
|
|
42
|
+
if select:
|
|
43
|
+
params["select"] = select
|
|
44
|
+
data = self.get("/works", params)
|
|
45
|
+
return (data or {}).get("message", {}).get("items", [])
|
|
46
|
+
|
|
47
|
+
def search_works(self, query: str, rows: int = 50,
|
|
48
|
+
from_year: Optional[int] = None) -> list[dict]:
|
|
49
|
+
params: dict = {"query.bibliographic": query, "rows": min(rows, 1000)}
|
|
50
|
+
if from_year:
|
|
51
|
+
params["filter"] = f"from-pub-date:{from_year}-01-01"
|
|
52
|
+
data = self.get("/works", params)
|
|
53
|
+
return (data or {}).get("message", {}).get("items", [])
|
|
54
|
+
|
|
55
|
+
def get_references(self, doi: str) -> list[dict]:
|
|
56
|
+
"""Only publishers who deposit references expose them."""
|
|
57
|
+
work = self.get_work_by_doi(doi)
|
|
58
|
+
return (work or {}).get("reference", []) or []
|
|
59
|
+
|
|
60
|
+
def get_abstract(self, doi: str) -> Optional[str]:
|
|
61
|
+
"""CrossRef abstracts are JATS-wrapped; strip the tags."""
|
|
62
|
+
import re
|
|
63
|
+
|
|
64
|
+
work = self.get_work_by_doi(doi)
|
|
65
|
+
raw = (work or {}).get("abstract")
|
|
66
|
+
if not raw:
|
|
67
|
+
return None
|
|
68
|
+
text = re.sub(r"<[^>]+>", " ", raw)
|
|
69
|
+
text = re.sub(r"\s+", " ", text).strip()
|
|
70
|
+
return text or None
|
|
71
|
+
|
|
72
|
+
def enrich_doi_metadata(self, doi: str) -> Optional[dict]:
|
|
73
|
+
"""Normalise a CrossRef record into a flat shape."""
|
|
74
|
+
work = self.get_work_by_doi(doi)
|
|
75
|
+
if not work:
|
|
76
|
+
return None
|
|
77
|
+
authors = []
|
|
78
|
+
for a in work.get("author", []) or []:
|
|
79
|
+
given, family = a.get("given"), a.get("family")
|
|
80
|
+
authors.append(" ".join(x for x in (given, family) if x) or a.get("name", ""))
|
|
81
|
+
issued = ((work.get("issued") or {}).get("date-parts") or [[None]])[0]
|
|
82
|
+
titles = work.get("title") or []
|
|
83
|
+
containers = work.get("container-title") or []
|
|
84
|
+
return {
|
|
85
|
+
"doi": work.get("DOI"),
|
|
86
|
+
"title": titles[0] if titles else None,
|
|
87
|
+
"authors": [a for a in authors if a],
|
|
88
|
+
"year": issued[0] if issued else None,
|
|
89
|
+
"journal": containers[0] if containers else None,
|
|
90
|
+
"type": work.get("type"),
|
|
91
|
+
"reference_count": work.get("references-count"),
|
|
92
|
+
"is_referenced_by_count": work.get("is-referenced-by-count"),
|
|
93
|
+
}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""OpenAIRE client. Free, no key.
|
|
2
|
+
|
|
3
|
+
Strong on European, multilingual and repository-held material, which matters
|
|
4
|
+
because a fifth of the JSTOR corpus is non-English. The response JSON is deeply
|
|
5
|
+
nested and inconsistently typed, so everything goes through _first().
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
from typing import Any, Optional
|
|
12
|
+
|
|
13
|
+
from scholarlib.http.base import BaseClient
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
BASE_URL = "https://api.openaire.eu/search"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _first(node: Any, *path: str) -> Any:
|
|
21
|
+
"""Walk a path, tolerating dict-or-list-of-dicts at every level."""
|
|
22
|
+
cur = node
|
|
23
|
+
for key in path:
|
|
24
|
+
if isinstance(cur, list):
|
|
25
|
+
cur = cur[0] if cur else None
|
|
26
|
+
if not isinstance(cur, dict):
|
|
27
|
+
return None
|
|
28
|
+
cur = cur.get(key)
|
|
29
|
+
if isinstance(cur, list):
|
|
30
|
+
cur = cur[0] if cur else None
|
|
31
|
+
return cur
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _text(node: Any) -> Optional[str]:
|
|
35
|
+
"""OpenAIRE wraps scalars as {'$': value}."""
|
|
36
|
+
if isinstance(node, list):
|
|
37
|
+
node = node[0] if node else None
|
|
38
|
+
if isinstance(node, dict):
|
|
39
|
+
node = node.get("$")
|
|
40
|
+
if node is None:
|
|
41
|
+
return None
|
|
42
|
+
s = str(node).strip()
|
|
43
|
+
return s or None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class OpenAIREClient(BaseClient):
|
|
47
|
+
name = "openaire"
|
|
48
|
+
base_url = BASE_URL
|
|
49
|
+
|
|
50
|
+
def _results(self, data: Optional[dict]) -> list[dict]:
|
|
51
|
+
res = _first(data, "response", "results")
|
|
52
|
+
if not isinstance(res, dict):
|
|
53
|
+
return []
|
|
54
|
+
rows = res.get("result")
|
|
55
|
+
if rows is None:
|
|
56
|
+
return []
|
|
57
|
+
return rows if isinstance(rows, list) else [rows]
|
|
58
|
+
|
|
59
|
+
def _entity(self, row: dict) -> Optional[dict]:
|
|
60
|
+
ent = _first(row, "metadata", "oaf:entity", "oaf:result")
|
|
61
|
+
return ent if isinstance(ent, dict) else None
|
|
62
|
+
|
|
63
|
+
def search(self, query: str, *, size: int = 20,
|
|
64
|
+
from_year: Optional[int] = None,
|
|
65
|
+
to_year: Optional[int] = None) -> list[dict]:
|
|
66
|
+
params: dict = {"keywords": query, "size": min(size, 100), "format": "json"}
|
|
67
|
+
if from_year:
|
|
68
|
+
params["fromDateAccepted"] = f"{from_year}-01-01"
|
|
69
|
+
if to_year:
|
|
70
|
+
params["toDateAccepted"] = f"{to_year}-12-31"
|
|
71
|
+
return self._results(self.get("/publications", params))
|
|
72
|
+
|
|
73
|
+
def get_by_doi(self, doi: str) -> Optional[dict]:
|
|
74
|
+
d = str(doi).replace("https://doi.org/", "").strip()
|
|
75
|
+
rows = self._results(self.get("/publications", {"doi": d, "format": "json", "size": 1}))
|
|
76
|
+
return rows[0] if rows else None
|
|
77
|
+
|
|
78
|
+
def get_abstract(self, doi: str) -> Optional[str]:
|
|
79
|
+
row = self.get_by_doi(doi)
|
|
80
|
+
if not row:
|
|
81
|
+
return None
|
|
82
|
+
ent = self._entity(row)
|
|
83
|
+
if not ent:
|
|
84
|
+
return None
|
|
85
|
+
text = _text(ent.get("description"))
|
|
86
|
+
if text and len(text) > 20:
|
|
87
|
+
return text
|
|
88
|
+
return None
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""OpenAlex client.
|
|
2
|
+
|
|
3
|
+
Credit costs measured 2026-09-20: singleton 0, list/cites/group_by 1,
|
|
4
|
+
batch of 50 ids 1, search 10. A free API key raises the daily budget from
|
|
5
|
+
~1000 to ~10000 credits.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
from typing import Iterable, Iterator, Optional
|
|
12
|
+
|
|
13
|
+
from scholarlib.http.base import BaseClient
|
|
14
|
+
from scholarlib.http.budget import COSTS, classify
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
BASE_URL = "https://api.openalex.org"
|
|
19
|
+
|
|
20
|
+
# Fields for researcher profiling (scholarfocus).
|
|
21
|
+
WORK_FIELDS = (
|
|
22
|
+
"id,doi,title,publication_year,type,"
|
|
23
|
+
"authorships,concepts,topics,keywords,"
|
|
24
|
+
"referenced_works,cited_by_count,"
|
|
25
|
+
"abstract_inverted_index"
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Minimal fields for bulk reference lookups.
|
|
29
|
+
REF_WORK_FIELDS = "id,doi,title,publication_year,authorships,cited_by_count"
|
|
30
|
+
|
|
31
|
+
# Literature review needs abstracts and the full citation edges.
|
|
32
|
+
# related_works matters because monographs have referenced_works_count == 0.
|
|
33
|
+
LITREVIEW_WORK_FIELDS = (
|
|
34
|
+
"id,doi,title,display_name,publication_year,type,language,"
|
|
35
|
+
"authorships,topics,concepts,keywords,"
|
|
36
|
+
"referenced_works,referenced_works_count,related_works,"
|
|
37
|
+
"cited_by_count,abstract_inverted_index,"
|
|
38
|
+
"primary_location,best_oa_location,open_access"
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
BATCH_SIZE = 50
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def reconstruct_abstract(inverted_index: Optional[dict]) -> Optional[str]:
|
|
45
|
+
"""Rebuild plain text from OpenAlex's inverted index.
|
|
46
|
+
|
|
47
|
+
~92% of works carry one, which is why Semantic Scholar is an
|
|
48
|
+
enhancement rather than a dependency.
|
|
49
|
+
"""
|
|
50
|
+
if not inverted_index:
|
|
51
|
+
return None
|
|
52
|
+
positions: list[tuple[int, str]] = []
|
|
53
|
+
for word, idxs in inverted_index.items():
|
|
54
|
+
for i in idxs:
|
|
55
|
+
positions.append((i, word))
|
|
56
|
+
if not positions:
|
|
57
|
+
return None
|
|
58
|
+
positions.sort()
|
|
59
|
+
return " ".join(word for _, word in positions)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def short_id(work_id: str) -> str:
|
|
63
|
+
"""https://openalex.org/W123 -> W123"""
|
|
64
|
+
return str(work_id).rstrip("/").rsplit("/", 1)[-1]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def chunked(items: Iterable, size: int) -> Iterator[list]:
|
|
68
|
+
buf: list = []
|
|
69
|
+
for it in items:
|
|
70
|
+
buf.append(it)
|
|
71
|
+
if len(buf) >= size:
|
|
72
|
+
yield buf
|
|
73
|
+
buf = []
|
|
74
|
+
if buf:
|
|
75
|
+
yield buf
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class OpenAlexClient(BaseClient):
|
|
79
|
+
name = "openalex"
|
|
80
|
+
base_url = BASE_URL
|
|
81
|
+
|
|
82
|
+
def __init__(self, email: Optional[str] = None, api_key: Optional[str] = None, **kw):
|
|
83
|
+
super().__init__(contact_email=email, **kw)
|
|
84
|
+
self.email = email
|
|
85
|
+
self.api_key = api_key
|
|
86
|
+
|
|
87
|
+
def _auth_params(self) -> dict:
|
|
88
|
+
p = {}
|
|
89
|
+
if self.email:
|
|
90
|
+
p["mailto"] = self.email
|
|
91
|
+
if self.api_key:
|
|
92
|
+
p["api_key"] = self.api_key
|
|
93
|
+
return p
|
|
94
|
+
|
|
95
|
+
def _cost(self, path: str, params: dict) -> tuple[str, int]:
|
|
96
|
+
klass = classify(path, params)
|
|
97
|
+
return klass, COSTS.get(klass, 1)
|
|
98
|
+
|
|
99
|
+
def _ttl(self, path: str, params: dict) -> int:
|
|
100
|
+
"""Cost drives TTL: searches are 10 credits, singletons are free."""
|
|
101
|
+
klass = classify(path, params)
|
|
102
|
+
if klass == "search":
|
|
103
|
+
return 90 * 86400
|
|
104
|
+
if klass == "singleton":
|
|
105
|
+
return 14 * 86400
|
|
106
|
+
return 30 * 86400
|
|
107
|
+
|
|
108
|
+
# ---- authors --------------------------------------------------------
|
|
109
|
+
|
|
110
|
+
def get_author_by_orcid(self, orcid: str) -> Optional[dict]:
|
|
111
|
+
o = str(orcid).strip()
|
|
112
|
+
if not o.startswith("http"):
|
|
113
|
+
o = f"https://orcid.org/{o}"
|
|
114
|
+
return self.get(f"/authors/{o}")
|
|
115
|
+
|
|
116
|
+
def search_authors(self, name: str, limit: int = 10) -> list[dict]:
|
|
117
|
+
data = self.get("/authors", {"search": name, "per-page": limit})
|
|
118
|
+
return (data or {}).get("results", [])
|
|
119
|
+
|
|
120
|
+
def get_author_by_id(self, author_id: str) -> Optional[dict]:
|
|
121
|
+
return self.get(f"/authors/{short_id(author_id)}")
|
|
122
|
+
|
|
123
|
+
# ---- works ----------------------------------------------------------
|
|
124
|
+
|
|
125
|
+
def get_work(self, work_id: str, select: Optional[str] = None) -> Optional[dict]:
|
|
126
|
+
"""Free: singleton lookups cost 0 credits."""
|
|
127
|
+
params = {"select": select} if select else None
|
|
128
|
+
return self.get(f"/works/{short_id(work_id)}", params)
|
|
129
|
+
|
|
130
|
+
def get_work_by_doi(self, doi: str, select: Optional[str] = None) -> Optional[dict]:
|
|
131
|
+
"""Free. Always prefer this over searching for a title."""
|
|
132
|
+
d = str(doi).replace("https://doi.org/", "").strip()
|
|
133
|
+
params = {"select": select} if select else None
|
|
134
|
+
return self.get(f"/works/doi:{d}", params)
|
|
135
|
+
|
|
136
|
+
def get_author_works(self, author_id: str, max_works: int = 100,
|
|
137
|
+
select: str = WORK_FIELDS) -> list[dict]:
|
|
138
|
+
return list(self.get_paged(
|
|
139
|
+
"/works",
|
|
140
|
+
{"filter": f"author.id:{short_id(author_id)}",
|
|
141
|
+
"select": select, "sort": "cited_by_count:desc"},
|
|
142
|
+
page_size=min(max_works, 200), max_items=max_works,
|
|
143
|
+
))
|
|
144
|
+
|
|
145
|
+
def get_works_batch(self, work_ids: Iterable[str],
|
|
146
|
+
select: str = REF_WORK_FIELDS) -> list[dict]:
|
|
147
|
+
"""50 works per credit — the cheapest way to hydrate known IDs."""
|
|
148
|
+
ids = [short_id(w) for w in work_ids if w]
|
|
149
|
+
out: list[dict] = []
|
|
150
|
+
for chunk in chunked(ids, BATCH_SIZE):
|
|
151
|
+
data = self.get("/works", {
|
|
152
|
+
"filter": f"ids.openalex:{'|'.join(chunk)}",
|
|
153
|
+
"per-page": BATCH_SIZE,
|
|
154
|
+
"select": select,
|
|
155
|
+
})
|
|
156
|
+
if data:
|
|
157
|
+
out.extend(data.get("results") or [])
|
|
158
|
+
return out
|
|
159
|
+
|
|
160
|
+
def search_works(self, query: str, *, limit: int = 50, from_year: Optional[int] = None,
|
|
161
|
+
to_year: Optional[int] = None, types: Optional[list[str]] = None,
|
|
162
|
+
languages: Optional[list[str]] = None,
|
|
163
|
+
select: str = LITREVIEW_WORK_FIELDS) -> list[dict]:
|
|
164
|
+
"""10 credits per page — the expensive operation. Use sparingly."""
|
|
165
|
+
filters = []
|
|
166
|
+
if from_year:
|
|
167
|
+
filters.append(f"from_publication_date:{from_year}-01-01")
|
|
168
|
+
if to_year:
|
|
169
|
+
filters.append(f"to_publication_date:{to_year}-12-31")
|
|
170
|
+
if types:
|
|
171
|
+
filters.append(f"type:{'|'.join(types)}")
|
|
172
|
+
if languages:
|
|
173
|
+
filters.append(f"language:{'|'.join(languages)}")
|
|
174
|
+
params = {"search": query, "per-page": min(limit, 200), "select": select}
|
|
175
|
+
if filters:
|
|
176
|
+
params["filter"] = ",".join(filters)
|
|
177
|
+
data = self.get("/works", params)
|
|
178
|
+
return (data or {}).get("results", [])
|
|
179
|
+
|
|
180
|
+
def cited_by(self, work_ids: Iterable[str], *, max_items: int = 200,
|
|
181
|
+
from_year: Optional[int] = None,
|
|
182
|
+
select: str = LITREVIEW_WORK_FIELDS) -> list[dict]:
|
|
183
|
+
"""Forward snowballing. The OR-packed filter makes 50 seeds cost 1 credit."""
|
|
184
|
+
ids = [short_id(w) for w in work_ids if w]
|
|
185
|
+
out: list[dict] = []
|
|
186
|
+
for chunk in chunked(ids, BATCH_SIZE):
|
|
187
|
+
filt = f"cites:{'|'.join(chunk)}"
|
|
188
|
+
if from_year:
|
|
189
|
+
filt += f",from_publication_date:{from_year}-01-01"
|
|
190
|
+
out.extend(self.get_paged(
|
|
191
|
+
"/works", {"filter": filt, "select": select},
|
|
192
|
+
page_size=200, max_items=max_items,
|
|
193
|
+
))
|
|
194
|
+
return out
|
|
195
|
+
|
|
196
|
+
def group_by(self, filter_expr: str, dimension: str) -> list[dict]:
|
|
197
|
+
"""Aggregate shape of a result set for 1 credit, fetching no works."""
|
|
198
|
+
data = self.get("/works", {"filter": filter_expr, "group_by": dimension})
|
|
199
|
+
return (data or {}).get("group_by", [])
|
scholarlib/apis/orcid.py
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""ORCID public API client.
|
|
2
|
+
|
|
3
|
+
Free, no key. The works list is *author-curated*, which makes it far more
|
|
4
|
+
complete than OpenAlex for researchers whose records OpenAlex has fragmented.
|
|
5
|
+
Tuomas Tammisto, for instance, has 50 works in ORCID and 1 in OpenAlex.
|
|
6
|
+
|
|
7
|
+
It carries no citation counts or topics, so it pairs with OpenAlex: ORCID
|
|
8
|
+
supplies the authoritative work list, OpenAlex hydrates the DOIs (free, since
|
|
9
|
+
singleton lookups cost 0 credits).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
from typing import Optional
|
|
16
|
+
|
|
17
|
+
from scholarlib.http.base import BaseClient
|
|
18
|
+
from scholarlib.records import Record
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
BASE_URL = "https://pub.orcid.org/v3.0"
|
|
23
|
+
|
|
24
|
+
_TYPE_MAP = {
|
|
25
|
+
"journal-article": "article",
|
|
26
|
+
"book": "book",
|
|
27
|
+
"book-chapter": "chapter",
|
|
28
|
+
"book-review": "review",
|
|
29
|
+
"conference-paper": "conference-paper",
|
|
30
|
+
"dissertation-thesis": "dissertation",
|
|
31
|
+
"preprint": "preprint",
|
|
32
|
+
"report": "report",
|
|
33
|
+
"edited-book": "book",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _val(node, *path):
|
|
38
|
+
cur = node
|
|
39
|
+
for k in path:
|
|
40
|
+
if not isinstance(cur, dict):
|
|
41
|
+
return None
|
|
42
|
+
cur = cur.get(k)
|
|
43
|
+
return cur
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class OrcidClient(BaseClient):
|
|
47
|
+
name = "orcid"
|
|
48
|
+
base_url = BASE_URL
|
|
49
|
+
default_ttl = 14 * 86400
|
|
50
|
+
|
|
51
|
+
def _auth_headers(self) -> dict:
|
|
52
|
+
return {"Accept": "application/json"}
|
|
53
|
+
|
|
54
|
+
def search_by_name(self, name: str, *, rows: int = 10) -> list[dict]:
|
|
55
|
+
"""Find ORCID iDs for a display name. Free, no key."""
|
|
56
|
+
parts = str(name).strip().split()
|
|
57
|
+
if len(parts) >= 2:
|
|
58
|
+
given, family = " ".join(parts[:-1]), parts[-1]
|
|
59
|
+
q = f'family-name:{family} AND given-names:{given}'
|
|
60
|
+
else:
|
|
61
|
+
q = f'family-name:{parts[0]}' if parts else ""
|
|
62
|
+
if not q:
|
|
63
|
+
return []
|
|
64
|
+
data = self.get("/expanded-search/", {"q": q, "rows": rows})
|
|
65
|
+
out = []
|
|
66
|
+
for r in (data or {}).get("expanded-result") or []:
|
|
67
|
+
out.append({
|
|
68
|
+
"orcid": r.get("orcid-id"),
|
|
69
|
+
"name": " ".join(x for x in (r.get("given-names"),
|
|
70
|
+
r.get("family-names")) if x),
|
|
71
|
+
"institutions": list(r.get("institution-name") or []),
|
|
72
|
+
})
|
|
73
|
+
return out
|
|
74
|
+
|
|
75
|
+
def get_person(self, orcid: str) -> Optional[dict]:
|
|
76
|
+
return self.get(f"/{_clean(orcid)}/person")
|
|
77
|
+
|
|
78
|
+
def get_employments(self, orcid: str) -> list[str]:
|
|
79
|
+
data = self.get(f"/{_clean(orcid)}/employments")
|
|
80
|
+
out = []
|
|
81
|
+
for grp in (data or {}).get("affiliation-group") or []:
|
|
82
|
+
for s in grp.get("summaries") or []:
|
|
83
|
+
nm = _val(s, "employment-summary", "organization", "name")
|
|
84
|
+
if nm and nm not in out:
|
|
85
|
+
out.append(nm)
|
|
86
|
+
return out
|
|
87
|
+
|
|
88
|
+
def get_works(self, orcid: str, *, max_works: int = 500) -> list[Record]:
|
|
89
|
+
"""The author-curated work list, as canonical Records."""
|
|
90
|
+
data = self.get(f"/{_clean(orcid)}/works")
|
|
91
|
+
groups = (data or {}).get("group") or []
|
|
92
|
+
out: list[Record] = []
|
|
93
|
+
for g in groups[:max_works]:
|
|
94
|
+
summaries = g.get("work-summary") or []
|
|
95
|
+
if not summaries:
|
|
96
|
+
continue
|
|
97
|
+
s = summaries[0]
|
|
98
|
+
title = _val(s, "title", "title", "value")
|
|
99
|
+
if not title:
|
|
100
|
+
continue
|
|
101
|
+
year = _val(s, "publication-date", "year", "value")
|
|
102
|
+
ids = {e.get("external-id-type"): e.get("external-id-value")
|
|
103
|
+
for e in (_val(g, "external-ids", "external-id") or [])}
|
|
104
|
+
out.append(Record(
|
|
105
|
+
doi=ids.get("doi"),
|
|
106
|
+
title=title,
|
|
107
|
+
year=int(year) if year and str(year).isdigit() else None,
|
|
108
|
+
venue=_val(s, "journal-title", "value"),
|
|
109
|
+
type=_TYPE_MAP.get(s.get("type"), s.get("type")),
|
|
110
|
+
provenance=["orcid"],
|
|
111
|
+
))
|
|
112
|
+
logger.info("ORCID %s: %d works (%d with DOIs)",
|
|
113
|
+
_clean(orcid), len(out), sum(1 for r in out if r.doi))
|
|
114
|
+
return out
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _clean(orcid: str) -> str:
|
|
118
|
+
return str(orcid).strip().replace("https://orcid.org/", "")
|