scholarlib 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. scholarlib/__init__.py +3 -0
  2. scholarlib/adapters.py +94 -0
  3. scholarlib/apis/__init__.py +0 -0
  4. scholarlib/apis/core_api.py +41 -0
  5. scholarlib/apis/crossref.py +93 -0
  6. scholarlib/apis/openaire.py +88 -0
  7. scholarlib/apis/openalex.py +199 -0
  8. scholarlib/apis/orcid.py +118 -0
  9. scholarlib/apis/semantic_scholar.py +99 -0
  10. scholarlib/apis/unpaywall.py +47 -0
  11. scholarlib/apis/zotero_local.py +156 -0
  12. scholarlib/cli/__init__.py +0 -0
  13. scholarlib/cli/jstor_index.py +165 -0
  14. scholarlib/cli/litreview.py +326 -0
  15. scholarlib/cli/scholarfocus.py +662 -0
  16. scholarlib/config.example.yaml +68 -0
  17. scholarlib/config.py +172 -0
  18. scholarlib/dedup.py +177 -0
  19. scholarlib/http/__init__.py +0 -0
  20. scholarlib/http/base.py +212 -0
  21. scholarlib/http/budget.py +162 -0
  22. scholarlib/http/cache.py +144 -0
  23. scholarlib/http/policy.py +59 -0
  24. scholarlib/jstor/__init__.py +0 -0
  25. scholarlib/jstor/build.py +336 -0
  26. scholarlib/jstor/indexes.sql +8 -0
  27. scholarlib/jstor/query.py +295 -0
  28. scholarlib/jstor/schema.sql +48 -0
  29. scholarlib/pipeline/__init__.py +0 -0
  30. scholarlib/pipeline/abstracts.py +59 -0
  31. scholarlib/pipeline/cluster.py +117 -0
  32. scholarlib/pipeline/context.py +92 -0
  33. scholarlib/pipeline/disambiguate.py +377 -0
  34. scholarlib/pipeline/jstor_coverage.py +137 -0
  35. scholarlib/pipeline/oa_links.py +61 -0
  36. scholarlib/pipeline/profile.py +333 -0
  37. scholarlib/pipeline/rank.py +139 -0
  38. scholarlib/pipeline/seed.py +97 -0
  39. scholarlib/pipeline/snowball.py +120 -0
  40. scholarlib/progress.py +131 -0
  41. scholarlib/records.py +202 -0
  42. scholarlib/render/__init__.py +0 -0
  43. scholarlib/render/bib.py +139 -0
  44. scholarlib/render/json_out.py +120 -0
  45. scholarlib/render/narrative.py +167 -0
  46. scholarlib/render/systematic.py +93 -0
  47. scholarlib-0.3.0.dist-info/METADATA +232 -0
  48. scholarlib-0.3.0.dist-info/RECORD +52 -0
  49. scholarlib-0.3.0.dist-info/WHEEL +5 -0
  50. scholarlib-0.3.0.dist-info/entry_points.txt +4 -0
  51. scholarlib-0.3.0.dist-info/licenses/LICENSE +21 -0
  52. scholarlib-0.3.0.dist-info/top_level.txt +1 -0
scholarlib/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Shared bibliographic-API library behind the scholarfocus and litreview skills."""
2
+
3
+ __version__ = "0.3.0"
scholarlib/adapters.py ADDED
@@ -0,0 +1,94 @@
1
+ """Convert each source's native shape into the canonical Record."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Optional
6
+
7
+ from scholarlib.apis.openalex import reconstruct_abstract
8
+ from scholarlib.records import Record
9
+
10
+
11
+ def _authors_from_openalex(work: dict) -> list[str]:
12
+ out = []
13
+ for a in work.get("authorships") or []:
14
+ name = ((a.get("author") or {}).get("display_name"))
15
+ if name:
16
+ out.append(name)
17
+ return out
18
+
19
+
20
+ def _topics_from_openalex(work: dict) -> list[str]:
21
+ names = []
22
+ for t in (work.get("topics") or [])[:5]:
23
+ if t.get("display_name"):
24
+ names.append(t["display_name"])
25
+ for c in (work.get("concepts") or [])[:5]:
26
+ if c.get("display_name") and c["display_name"] not in names:
27
+ names.append(c["display_name"])
28
+ return names
29
+
30
+
31
+ def from_openalex_work(work: dict, *, provenance: Optional[str] = None) -> Record:
32
+ abstract = reconstruct_abstract(work.get("abstract_inverted_index"))
33
+ loc = work.get("primary_location") or {}
34
+ src = loc.get("source") or {}
35
+ best_oa = work.get("best_oa_location") or {}
36
+ oa = work.get("open_access") or {}
37
+ return Record(
38
+ doi=work.get("doi"),
39
+ openalex_id=work.get("id"),
40
+ title=work.get("title") or work.get("display_name"),
41
+ authors=_authors_from_openalex(work),
42
+ year=work.get("publication_year"),
43
+ venue=src.get("display_name"),
44
+ type=work.get("type"),
45
+ language=work.get("language"),
46
+ abstract=abstract,
47
+ abstract_source="openalex" if abstract else None,
48
+ cited_by_count=work.get("cited_by_count"),
49
+ referenced_works=list(work.get("referenced_works") or []),
50
+ related_works=list(work.get("related_works") or []),
51
+ topics=_topics_from_openalex(work),
52
+ oa_status=oa.get("oa_status"),
53
+ oa_url=best_oa.get("pdf_url") or best_oa.get("landing_page_url"),
54
+ provenance=[provenance] if provenance else [],
55
+ )
56
+
57
+
58
+ def from_crossref_item(item: dict, *, provenance: str = "crossref") -> Record:
59
+ titles = item.get("title") or []
60
+ containers = item.get("container-title") or []
61
+ authors = []
62
+ for a in item.get("author") or []:
63
+ nm = " ".join(x for x in (a.get("given"), a.get("family")) if x) or a.get("name")
64
+ if nm:
65
+ authors.append(nm)
66
+ issued = ((item.get("issued") or {}).get("date-parts") or [[None]])[0]
67
+ return Record(
68
+ doi=item.get("DOI"),
69
+ title=titles[0] if titles else None,
70
+ authors=authors,
71
+ year=issued[0] if issued else None,
72
+ venue=containers[0] if containers else None,
73
+ type=item.get("type"),
74
+ cited_by_count=item.get("is-referenced-by-count"),
75
+ provenance=[provenance],
76
+ )
77
+
78
+
79
+ def from_s2_paper(paper: dict, *, provenance: str = "semantic_scholar") -> Record:
80
+ ext = paper.get("externalIds") or {}
81
+ tldr = (paper.get("tldr") or {}).get("text")
82
+ abstract = paper.get("abstract") or tldr
83
+ return Record(
84
+ doi=ext.get("DOI"),
85
+ title=paper.get("title"),
86
+ year=paper.get("year"),
87
+ venue=paper.get("venue"),
88
+ abstract=abstract,
89
+ abstract_source=("semantic_scholar" if paper.get("abstract")
90
+ else ("s2_tldr" if tldr else None)),
91
+ cited_by_count=paper.get("citationCount"),
92
+ topics=list(paper.get("fieldsOfStudy") or []),
93
+ provenance=[provenance],
94
+ )
File without changes
@@ -0,0 +1,41 @@
1
+ """CORE client. Key is optional but raises the rate limit from 10/min to 150/min."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from typing import Optional
7
+
8
+ from scholarlib.http.base import BaseClient
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+ BASE_URL = "https://api.core.ac.uk/v3"
13
+
14
+
15
+ class COREClient(BaseClient):
16
+ name = "core"
17
+ base_url = BASE_URL
18
+
19
+ def __init__(self, api_key: Optional[str] = None, **kw):
20
+ super().__init__(**kw)
21
+ self.api_key = api_key
22
+ if not api_key:
23
+ logger.debug("CORE: no API key — rate limit is 10/min instead of 150/min")
24
+
25
+ def _auth_headers(self) -> dict:
26
+ return {"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}
27
+
28
+ def search_works(self, query: str, limit: int = 10) -> list[dict]:
29
+ data = self.get("/search/works", {"q": query, "limit": limit})
30
+ return (data or {}).get("results", [])
31
+
32
+ def get_work_by_doi(self, doi: str) -> Optional[dict]:
33
+ d = str(doi).replace("https://doi.org/", "").strip()
34
+ results = self.search_works(f'doi:"{d}"', limit=1)
35
+ return results[0] if results else None
36
+
37
+ def get_abstract(self, doi: str) -> Optional[str]:
38
+ w = self.get_work_by_doi(doi)
39
+ if not w:
40
+ return None
41
+ return w.get("abstract") or w.get("description") or None
@@ -0,0 +1,93 @@
1
+ """CrossRef client. Free, polite pool via a mailto in the User-Agent."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from typing import Optional
7
+
8
+ from scholarlib.http.base import BaseClient
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+ BASE_URL = "https://api.crossref.org"
13
+
14
+ # Verified against the live API. `reference-count` is NOT a valid select field
15
+ # and returns HTTP 400 select-not-available -- that was the long-standing bug.
16
+ # The valid spellings are `references-count` and `is-referenced-by-count`.
17
+ SAFE_SELECT = (
18
+ "DOI,title,author,issued,type,container-title,"
19
+ "references-count,is-referenced-by-count,abstract"
20
+ )
21
+
22
+
23
+ class CrossRefClient(BaseClient):
24
+ name = "crossref"
25
+ base_url = BASE_URL
26
+
27
+ def __init__(self, email: Optional[str] = None, **kw):
28
+ super().__init__(contact_email=email, **kw)
29
+ self.email = email
30
+
31
+ def _auth_params(self) -> dict:
32
+ return {"mailto": self.email} if self.email else {}
33
+
34
+ def get_work_by_doi(self, doi: str) -> Optional[dict]:
35
+ d = str(doi).replace("https://doi.org/", "").strip()
36
+ data = self.get(f"/works/{d}")
37
+ return (data or {}).get("message")
38
+
39
+ def search_works_by_author(self, name: str, rows: int = 50,
40
+ *, select: Optional[str] = None) -> list[dict]:
41
+ params: dict = {"query.author": name, "rows": min(rows, 1000)}
42
+ if select:
43
+ params["select"] = select
44
+ data = self.get("/works", params)
45
+ return (data or {}).get("message", {}).get("items", [])
46
+
47
+ def search_works(self, query: str, rows: int = 50,
48
+ from_year: Optional[int] = None) -> list[dict]:
49
+ params: dict = {"query.bibliographic": query, "rows": min(rows, 1000)}
50
+ if from_year:
51
+ params["filter"] = f"from-pub-date:{from_year}-01-01"
52
+ data = self.get("/works", params)
53
+ return (data or {}).get("message", {}).get("items", [])
54
+
55
+ def get_references(self, doi: str) -> list[dict]:
56
+ """Only publishers who deposit references expose them."""
57
+ work = self.get_work_by_doi(doi)
58
+ return (work or {}).get("reference", []) or []
59
+
60
+ def get_abstract(self, doi: str) -> Optional[str]:
61
+ """CrossRef abstracts are JATS-wrapped; strip the tags."""
62
+ import re
63
+
64
+ work = self.get_work_by_doi(doi)
65
+ raw = (work or {}).get("abstract")
66
+ if not raw:
67
+ return None
68
+ text = re.sub(r"<[^>]+>", " ", raw)
69
+ text = re.sub(r"\s+", " ", text).strip()
70
+ return text or None
71
+
72
+ def enrich_doi_metadata(self, doi: str) -> Optional[dict]:
73
+ """Normalise a CrossRef record into a flat shape."""
74
+ work = self.get_work_by_doi(doi)
75
+ if not work:
76
+ return None
77
+ authors = []
78
+ for a in work.get("author", []) or []:
79
+ given, family = a.get("given"), a.get("family")
80
+ authors.append(" ".join(x for x in (given, family) if x) or a.get("name", ""))
81
+ issued = ((work.get("issued") or {}).get("date-parts") or [[None]])[0]
82
+ titles = work.get("title") or []
83
+ containers = work.get("container-title") or []
84
+ return {
85
+ "doi": work.get("DOI"),
86
+ "title": titles[0] if titles else None,
87
+ "authors": [a for a in authors if a],
88
+ "year": issued[0] if issued else None,
89
+ "journal": containers[0] if containers else None,
90
+ "type": work.get("type"),
91
+ "reference_count": work.get("references-count"),
92
+ "is_referenced_by_count": work.get("is-referenced-by-count"),
93
+ }
@@ -0,0 +1,88 @@
1
+ """OpenAIRE client. Free, no key.
2
+
3
+ Strong on European, multilingual and repository-held material, which matters
4
+ because a fifth of the JSTOR corpus is non-English. The response JSON is deeply
5
+ nested and inconsistently typed, so everything goes through _first().
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import logging
11
+ from typing import Any, Optional
12
+
13
+ from scholarlib.http.base import BaseClient
14
+
15
+ logger = logging.getLogger(__name__)
16
+
17
+ BASE_URL = "https://api.openaire.eu/search"
18
+
19
+
20
+ def _first(node: Any, *path: str) -> Any:
21
+ """Walk a path, tolerating dict-or-list-of-dicts at every level."""
22
+ cur = node
23
+ for key in path:
24
+ if isinstance(cur, list):
25
+ cur = cur[0] if cur else None
26
+ if not isinstance(cur, dict):
27
+ return None
28
+ cur = cur.get(key)
29
+ if isinstance(cur, list):
30
+ cur = cur[0] if cur else None
31
+ return cur
32
+
33
+
34
+ def _text(node: Any) -> Optional[str]:
35
+ """OpenAIRE wraps scalars as {'$': value}."""
36
+ if isinstance(node, list):
37
+ node = node[0] if node else None
38
+ if isinstance(node, dict):
39
+ node = node.get("$")
40
+ if node is None:
41
+ return None
42
+ s = str(node).strip()
43
+ return s or None
44
+
45
+
46
+ class OpenAIREClient(BaseClient):
47
+ name = "openaire"
48
+ base_url = BASE_URL
49
+
50
+ def _results(self, data: Optional[dict]) -> list[dict]:
51
+ res = _first(data, "response", "results")
52
+ if not isinstance(res, dict):
53
+ return []
54
+ rows = res.get("result")
55
+ if rows is None:
56
+ return []
57
+ return rows if isinstance(rows, list) else [rows]
58
+
59
+ def _entity(self, row: dict) -> Optional[dict]:
60
+ ent = _first(row, "metadata", "oaf:entity", "oaf:result")
61
+ return ent if isinstance(ent, dict) else None
62
+
63
+ def search(self, query: str, *, size: int = 20,
64
+ from_year: Optional[int] = None,
65
+ to_year: Optional[int] = None) -> list[dict]:
66
+ params: dict = {"keywords": query, "size": min(size, 100), "format": "json"}
67
+ if from_year:
68
+ params["fromDateAccepted"] = f"{from_year}-01-01"
69
+ if to_year:
70
+ params["toDateAccepted"] = f"{to_year}-12-31"
71
+ return self._results(self.get("/publications", params))
72
+
73
+ def get_by_doi(self, doi: str) -> Optional[dict]:
74
+ d = str(doi).replace("https://doi.org/", "").strip()
75
+ rows = self._results(self.get("/publications", {"doi": d, "format": "json", "size": 1}))
76
+ return rows[0] if rows else None
77
+
78
+ def get_abstract(self, doi: str) -> Optional[str]:
79
+ row = self.get_by_doi(doi)
80
+ if not row:
81
+ return None
82
+ ent = self._entity(row)
83
+ if not ent:
84
+ return None
85
+ text = _text(ent.get("description"))
86
+ if text and len(text) > 20:
87
+ return text
88
+ return None
@@ -0,0 +1,199 @@
1
+ """OpenAlex client.
2
+
3
+ Credit costs measured 2026-09-20: singleton 0, list/cites/group_by 1,
4
+ batch of 50 ids 1, search 10. A free API key raises the daily budget from
5
+ ~1000 to ~10000 credits.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import logging
11
+ from typing import Iterable, Iterator, Optional
12
+
13
+ from scholarlib.http.base import BaseClient
14
+ from scholarlib.http.budget import COSTS, classify
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ BASE_URL = "https://api.openalex.org"
19
+
20
+ # Fields for researcher profiling (scholarfocus).
21
+ WORK_FIELDS = (
22
+ "id,doi,title,publication_year,type,"
23
+ "authorships,concepts,topics,keywords,"
24
+ "referenced_works,cited_by_count,"
25
+ "abstract_inverted_index"
26
+ )
27
+
28
+ # Minimal fields for bulk reference lookups.
29
+ REF_WORK_FIELDS = "id,doi,title,publication_year,authorships,cited_by_count"
30
+
31
+ # Literature review needs abstracts and the full citation edges.
32
+ # related_works matters because monographs have referenced_works_count == 0.
33
+ LITREVIEW_WORK_FIELDS = (
34
+ "id,doi,title,display_name,publication_year,type,language,"
35
+ "authorships,topics,concepts,keywords,"
36
+ "referenced_works,referenced_works_count,related_works,"
37
+ "cited_by_count,abstract_inverted_index,"
38
+ "primary_location,best_oa_location,open_access"
39
+ )
40
+
41
+ BATCH_SIZE = 50
42
+
43
+
44
+ def reconstruct_abstract(inverted_index: Optional[dict]) -> Optional[str]:
45
+ """Rebuild plain text from OpenAlex's inverted index.
46
+
47
+ ~92% of works carry one, which is why Semantic Scholar is an
48
+ enhancement rather than a dependency.
49
+ """
50
+ if not inverted_index:
51
+ return None
52
+ positions: list[tuple[int, str]] = []
53
+ for word, idxs in inverted_index.items():
54
+ for i in idxs:
55
+ positions.append((i, word))
56
+ if not positions:
57
+ return None
58
+ positions.sort()
59
+ return " ".join(word for _, word in positions)
60
+
61
+
62
+ def short_id(work_id: str) -> str:
63
+ """https://openalex.org/W123 -> W123"""
64
+ return str(work_id).rstrip("/").rsplit("/", 1)[-1]
65
+
66
+
67
+ def chunked(items: Iterable, size: int) -> Iterator[list]:
68
+ buf: list = []
69
+ for it in items:
70
+ buf.append(it)
71
+ if len(buf) >= size:
72
+ yield buf
73
+ buf = []
74
+ if buf:
75
+ yield buf
76
+
77
+
78
+ class OpenAlexClient(BaseClient):
79
+ name = "openalex"
80
+ base_url = BASE_URL
81
+
82
+ def __init__(self, email: Optional[str] = None, api_key: Optional[str] = None, **kw):
83
+ super().__init__(contact_email=email, **kw)
84
+ self.email = email
85
+ self.api_key = api_key
86
+
87
+ def _auth_params(self) -> dict:
88
+ p = {}
89
+ if self.email:
90
+ p["mailto"] = self.email
91
+ if self.api_key:
92
+ p["api_key"] = self.api_key
93
+ return p
94
+
95
+ def _cost(self, path: str, params: dict) -> tuple[str, int]:
96
+ klass = classify(path, params)
97
+ return klass, COSTS.get(klass, 1)
98
+
99
+ def _ttl(self, path: str, params: dict) -> int:
100
+ """Cost drives TTL: searches are 10 credits, singletons are free."""
101
+ klass = classify(path, params)
102
+ if klass == "search":
103
+ return 90 * 86400
104
+ if klass == "singleton":
105
+ return 14 * 86400
106
+ return 30 * 86400
107
+
108
+ # ---- authors --------------------------------------------------------
109
+
110
+ def get_author_by_orcid(self, orcid: str) -> Optional[dict]:
111
+ o = str(orcid).strip()
112
+ if not o.startswith("http"):
113
+ o = f"https://orcid.org/{o}"
114
+ return self.get(f"/authors/{o}")
115
+
116
+ def search_authors(self, name: str, limit: int = 10) -> list[dict]:
117
+ data = self.get("/authors", {"search": name, "per-page": limit})
118
+ return (data or {}).get("results", [])
119
+
120
+ def get_author_by_id(self, author_id: str) -> Optional[dict]:
121
+ return self.get(f"/authors/{short_id(author_id)}")
122
+
123
+ # ---- works ----------------------------------------------------------
124
+
125
+ def get_work(self, work_id: str, select: Optional[str] = None) -> Optional[dict]:
126
+ """Free: singleton lookups cost 0 credits."""
127
+ params = {"select": select} if select else None
128
+ return self.get(f"/works/{short_id(work_id)}", params)
129
+
130
+ def get_work_by_doi(self, doi: str, select: Optional[str] = None) -> Optional[dict]:
131
+ """Free. Always prefer this over searching for a title."""
132
+ d = str(doi).replace("https://doi.org/", "").strip()
133
+ params = {"select": select} if select else None
134
+ return self.get(f"/works/doi:{d}", params)
135
+
136
+ def get_author_works(self, author_id: str, max_works: int = 100,
137
+ select: str = WORK_FIELDS) -> list[dict]:
138
+ return list(self.get_paged(
139
+ "/works",
140
+ {"filter": f"author.id:{short_id(author_id)}",
141
+ "select": select, "sort": "cited_by_count:desc"},
142
+ page_size=min(max_works, 200), max_items=max_works,
143
+ ))
144
+
145
+ def get_works_batch(self, work_ids: Iterable[str],
146
+ select: str = REF_WORK_FIELDS) -> list[dict]:
147
+ """50 works per credit — the cheapest way to hydrate known IDs."""
148
+ ids = [short_id(w) for w in work_ids if w]
149
+ out: list[dict] = []
150
+ for chunk in chunked(ids, BATCH_SIZE):
151
+ data = self.get("/works", {
152
+ "filter": f"ids.openalex:{'|'.join(chunk)}",
153
+ "per-page": BATCH_SIZE,
154
+ "select": select,
155
+ })
156
+ if data:
157
+ out.extend(data.get("results") or [])
158
+ return out
159
+
160
+ def search_works(self, query: str, *, limit: int = 50, from_year: Optional[int] = None,
161
+ to_year: Optional[int] = None, types: Optional[list[str]] = None,
162
+ languages: Optional[list[str]] = None,
163
+ select: str = LITREVIEW_WORK_FIELDS) -> list[dict]:
164
+ """10 credits per page — the expensive operation. Use sparingly."""
165
+ filters = []
166
+ if from_year:
167
+ filters.append(f"from_publication_date:{from_year}-01-01")
168
+ if to_year:
169
+ filters.append(f"to_publication_date:{to_year}-12-31")
170
+ if types:
171
+ filters.append(f"type:{'|'.join(types)}")
172
+ if languages:
173
+ filters.append(f"language:{'|'.join(languages)}")
174
+ params = {"search": query, "per-page": min(limit, 200), "select": select}
175
+ if filters:
176
+ params["filter"] = ",".join(filters)
177
+ data = self.get("/works", params)
178
+ return (data or {}).get("results", [])
179
+
180
+ def cited_by(self, work_ids: Iterable[str], *, max_items: int = 200,
181
+ from_year: Optional[int] = None,
182
+ select: str = LITREVIEW_WORK_FIELDS) -> list[dict]:
183
+ """Forward snowballing. The OR-packed filter makes 50 seeds cost 1 credit."""
184
+ ids = [short_id(w) for w in work_ids if w]
185
+ out: list[dict] = []
186
+ for chunk in chunked(ids, BATCH_SIZE):
187
+ filt = f"cites:{'|'.join(chunk)}"
188
+ if from_year:
189
+ filt += f",from_publication_date:{from_year}-01-01"
190
+ out.extend(self.get_paged(
191
+ "/works", {"filter": filt, "select": select},
192
+ page_size=200, max_items=max_items,
193
+ ))
194
+ return out
195
+
196
+ def group_by(self, filter_expr: str, dimension: str) -> list[dict]:
197
+ """Aggregate shape of a result set for 1 credit, fetching no works."""
198
+ data = self.get("/works", {"filter": filter_expr, "group_by": dimension})
199
+ return (data or {}).get("group_by", [])
@@ -0,0 +1,118 @@
1
+ """ORCID public API client.
2
+
3
+ Free, no key. The works list is *author-curated*, which makes it far more
4
+ complete than OpenAlex for researchers whose records OpenAlex has fragmented.
5
+ Tuomas Tammisto, for instance, has 50 works in ORCID and 1 in OpenAlex.
6
+
7
+ It carries no citation counts or topics, so it pairs with OpenAlex: ORCID
8
+ supplies the authoritative work list, OpenAlex hydrates the DOIs (free, since
9
+ singleton lookups cost 0 credits).
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import logging
15
+ from typing import Optional
16
+
17
+ from scholarlib.http.base import BaseClient
18
+ from scholarlib.records import Record
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+ BASE_URL = "https://pub.orcid.org/v3.0"
23
+
24
+ _TYPE_MAP = {
25
+ "journal-article": "article",
26
+ "book": "book",
27
+ "book-chapter": "chapter",
28
+ "book-review": "review",
29
+ "conference-paper": "conference-paper",
30
+ "dissertation-thesis": "dissertation",
31
+ "preprint": "preprint",
32
+ "report": "report",
33
+ "edited-book": "book",
34
+ }
35
+
36
+
37
+ def _val(node, *path):
38
+ cur = node
39
+ for k in path:
40
+ if not isinstance(cur, dict):
41
+ return None
42
+ cur = cur.get(k)
43
+ return cur
44
+
45
+
46
+ class OrcidClient(BaseClient):
47
+ name = "orcid"
48
+ base_url = BASE_URL
49
+ default_ttl = 14 * 86400
50
+
51
+ def _auth_headers(self) -> dict:
52
+ return {"Accept": "application/json"}
53
+
54
+ def search_by_name(self, name: str, *, rows: int = 10) -> list[dict]:
55
+ """Find ORCID iDs for a display name. Free, no key."""
56
+ parts = str(name).strip().split()
57
+ if len(parts) >= 2:
58
+ given, family = " ".join(parts[:-1]), parts[-1]
59
+ q = f'family-name:{family} AND given-names:{given}'
60
+ else:
61
+ q = f'family-name:{parts[0]}' if parts else ""
62
+ if not q:
63
+ return []
64
+ data = self.get("/expanded-search/", {"q": q, "rows": rows})
65
+ out = []
66
+ for r in (data or {}).get("expanded-result") or []:
67
+ out.append({
68
+ "orcid": r.get("orcid-id"),
69
+ "name": " ".join(x for x in (r.get("given-names"),
70
+ r.get("family-names")) if x),
71
+ "institutions": list(r.get("institution-name") or []),
72
+ })
73
+ return out
74
+
75
+ def get_person(self, orcid: str) -> Optional[dict]:
76
+ return self.get(f"/{_clean(orcid)}/person")
77
+
78
+ def get_employments(self, orcid: str) -> list[str]:
79
+ data = self.get(f"/{_clean(orcid)}/employments")
80
+ out = []
81
+ for grp in (data or {}).get("affiliation-group") or []:
82
+ for s in grp.get("summaries") or []:
83
+ nm = _val(s, "employment-summary", "organization", "name")
84
+ if nm and nm not in out:
85
+ out.append(nm)
86
+ return out
87
+
88
+ def get_works(self, orcid: str, *, max_works: int = 500) -> list[Record]:
89
+ """The author-curated work list, as canonical Records."""
90
+ data = self.get(f"/{_clean(orcid)}/works")
91
+ groups = (data or {}).get("group") or []
92
+ out: list[Record] = []
93
+ for g in groups[:max_works]:
94
+ summaries = g.get("work-summary") or []
95
+ if not summaries:
96
+ continue
97
+ s = summaries[0]
98
+ title = _val(s, "title", "title", "value")
99
+ if not title:
100
+ continue
101
+ year = _val(s, "publication-date", "year", "value")
102
+ ids = {e.get("external-id-type"): e.get("external-id-value")
103
+ for e in (_val(g, "external-ids", "external-id") or [])}
104
+ out.append(Record(
105
+ doi=ids.get("doi"),
106
+ title=title,
107
+ year=int(year) if year and str(year).isdigit() else None,
108
+ venue=_val(s, "journal-title", "value"),
109
+ type=_TYPE_MAP.get(s.get("type"), s.get("type")),
110
+ provenance=["orcid"],
111
+ ))
112
+ logger.info("ORCID %s: %d works (%d with DOIs)",
113
+ _clean(orcid), len(out), sum(1 for r in out if r.doi))
114
+ return out
115
+
116
+
117
+ def _clean(orcid: str) -> str:
118
+ return str(orcid).strip().replace("https://orcid.org/", "")