scholarlib 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. scholarlib-0.3.0/LICENSE +21 -0
  2. scholarlib-0.3.0/PKG-INFO +232 -0
  3. scholarlib-0.3.0/README.md +201 -0
  4. scholarlib-0.3.0/pyproject.toml +57 -0
  5. scholarlib-0.3.0/scholarlib/__init__.py +3 -0
  6. scholarlib-0.3.0/scholarlib/adapters.py +94 -0
  7. scholarlib-0.3.0/scholarlib/apis/__init__.py +0 -0
  8. scholarlib-0.3.0/scholarlib/apis/core_api.py +41 -0
  9. scholarlib-0.3.0/scholarlib/apis/crossref.py +93 -0
  10. scholarlib-0.3.0/scholarlib/apis/openaire.py +88 -0
  11. scholarlib-0.3.0/scholarlib/apis/openalex.py +199 -0
  12. scholarlib-0.3.0/scholarlib/apis/orcid.py +118 -0
  13. scholarlib-0.3.0/scholarlib/apis/semantic_scholar.py +99 -0
  14. scholarlib-0.3.0/scholarlib/apis/unpaywall.py +47 -0
  15. scholarlib-0.3.0/scholarlib/apis/zotero_local.py +156 -0
  16. scholarlib-0.3.0/scholarlib/cli/__init__.py +0 -0
  17. scholarlib-0.3.0/scholarlib/cli/jstor_index.py +165 -0
  18. scholarlib-0.3.0/scholarlib/cli/litreview.py +326 -0
  19. scholarlib-0.3.0/scholarlib/cli/scholarfocus.py +662 -0
  20. scholarlib-0.3.0/scholarlib/config.example.yaml +68 -0
  21. scholarlib-0.3.0/scholarlib/config.py +172 -0
  22. scholarlib-0.3.0/scholarlib/dedup.py +177 -0
  23. scholarlib-0.3.0/scholarlib/http/__init__.py +0 -0
  24. scholarlib-0.3.0/scholarlib/http/base.py +212 -0
  25. scholarlib-0.3.0/scholarlib/http/budget.py +162 -0
  26. scholarlib-0.3.0/scholarlib/http/cache.py +144 -0
  27. scholarlib-0.3.0/scholarlib/http/policy.py +59 -0
  28. scholarlib-0.3.0/scholarlib/jstor/__init__.py +0 -0
  29. scholarlib-0.3.0/scholarlib/jstor/build.py +336 -0
  30. scholarlib-0.3.0/scholarlib/jstor/indexes.sql +8 -0
  31. scholarlib-0.3.0/scholarlib/jstor/query.py +295 -0
  32. scholarlib-0.3.0/scholarlib/jstor/schema.sql +48 -0
  33. scholarlib-0.3.0/scholarlib/pipeline/__init__.py +0 -0
  34. scholarlib-0.3.0/scholarlib/pipeline/abstracts.py +59 -0
  35. scholarlib-0.3.0/scholarlib/pipeline/cluster.py +117 -0
  36. scholarlib-0.3.0/scholarlib/pipeline/context.py +92 -0
  37. scholarlib-0.3.0/scholarlib/pipeline/disambiguate.py +377 -0
  38. scholarlib-0.3.0/scholarlib/pipeline/jstor_coverage.py +137 -0
  39. scholarlib-0.3.0/scholarlib/pipeline/oa_links.py +61 -0
  40. scholarlib-0.3.0/scholarlib/pipeline/profile.py +333 -0
  41. scholarlib-0.3.0/scholarlib/pipeline/rank.py +139 -0
  42. scholarlib-0.3.0/scholarlib/pipeline/seed.py +97 -0
  43. scholarlib-0.3.0/scholarlib/pipeline/snowball.py +120 -0
  44. scholarlib-0.3.0/scholarlib/progress.py +131 -0
  45. scholarlib-0.3.0/scholarlib/records.py +202 -0
  46. scholarlib-0.3.0/scholarlib/render/__init__.py +0 -0
  47. scholarlib-0.3.0/scholarlib/render/bib.py +139 -0
  48. scholarlib-0.3.0/scholarlib/render/json_out.py +120 -0
  49. scholarlib-0.3.0/scholarlib/render/narrative.py +167 -0
  50. scholarlib-0.3.0/scholarlib/render/systematic.py +93 -0
  51. scholarlib-0.3.0/scholarlib.egg-info/PKG-INFO +232 -0
  52. scholarlib-0.3.0/scholarlib.egg-info/SOURCES.txt +59 -0
  53. scholarlib-0.3.0/scholarlib.egg-info/dependency_links.txt +1 -0
  54. scholarlib-0.3.0/scholarlib.egg-info/entry_points.txt +4 -0
  55. scholarlib-0.3.0/scholarlib.egg-info/requires.txt +11 -0
  56. scholarlib-0.3.0/scholarlib.egg-info/top_level.txt +1 -0
  57. scholarlib-0.3.0/setup.cfg +4 -0
  58. scholarlib-0.3.0/tests/test_budget.py +46 -0
  59. scholarlib-0.3.0/tests/test_dedup.py +155 -0
  60. scholarlib-0.3.0/tests/test_disambiguate.py +90 -0
  61. scholarlib-0.3.0/tests/test_jstor_query.py +29 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Heikki Wilenius
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,232 @@
1
+ Metadata-Version: 2.4
2
+ Name: scholarlib
3
+ Version: 0.3.0
4
+ Summary: Shared bibliographic-API library behind the scholarfocus and litreview agent skills
5
+ Author: Heikki Wilenius
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/wilenius/scholarfocus-skill
8
+ Project-URL: Repository, https://github.com/wilenius/scholarfocus-skill
9
+ Project-URL: Issues, https://github.com/wilenius/scholarfocus-skill/issues
10
+ Keywords: bibliography,openalex,orcid,literature-review,agent-skills
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
18
+ Classifier: Topic :: Text Processing :: Indexing
19
+ Requires-Python: >=3.11
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: requests>=2.31.0
23
+ Requires-Dist: pyyaml>=6.0
24
+ Provides-Extra: fast
25
+ Requires-Dist: rapidfuzz>=3.6; extra == "fast"
26
+ Provides-Extra: zotero
27
+ Requires-Dist: pyzotero>=1.5; extra == "zotero"
28
+ Provides-Extra: dev
29
+ Requires-Dist: pytest>=8.0; extra == "dev"
30
+ Dynamic: license-file
31
+
32
+ # scholarfocus-skill
33
+
34
+ Two agent skills over one shared bibliographic library, for Claude Code, OpenClaw
35
+ and any other assistant that can run a shell command.
36
+
37
+ | Skill | Object | Question it answers |
38
+ |---|---|---|
39
+ | [`skills/scholarfocus`](skills/scholarfocus/SKILL.md) | **people** | What does this researcher work on? Who with? What do they cite? |
40
+ | [`skills/litreview`](skills/litreview/SKILL.md) | **topics** | What is the state of the field on X? What are the key works? |
41
+
42
+ Both sit on [`scholarlib`](https://pypi.org/project/scholarlib/), which wraps OpenAlex,
43
+ ORCID, CrossRef, CORE, Unpaywall, OpenAIRE, Semantic Scholar and an optional local
44
+ JSTOR metadata index behind one cache, one credit budget and one deduplication layer.
45
+
46
+ ## Install
47
+
48
+ Two steps: the command-line tool, then the skills.
49
+
50
+ ### 1. The tool
51
+
52
+ Install `scholarlib` as a standalone tool, not into your system Python — on Arch,
53
+ Manjaro, Debian, Ubuntu and Fedora the system interpreter is marked
54
+ externally-managed and a bare `pip install` will refuse with *"This environment is
55
+ externally managed"*. That refusal is correct; do not override it.
56
+
57
+ **macOS**
58
+
59
+ ```bash
60
+ brew install uv # or: brew install pipx && pipx ensurepath
61
+ uv tool install scholarlib
62
+ ```
63
+
64
+ **Linux**
65
+
66
+ ```bash
67
+ # Arch / Manjaro
68
+ sudo pacman -S uv # or: sudo pacman -S python-pipx
69
+
70
+ # Debian / Ubuntu
71
+ sudo apt install pipx # then use `pipx install` below
72
+
73
+ # Fedora
74
+ sudo dnf install uv
75
+
76
+ uv tool install scholarlib # or: pipx install scholarlib
77
+ ```
78
+
79
+ No `uv` or `pipx` available? `python3 -m venv ~/.venvs/scholarlib && ~/.venvs/scholarlib/bin/pip install scholarlib`
80
+ works anywhere, and the commands land in `~/.venvs/scholarlib/bin/`.
81
+
82
+ Verify, and create a config:
83
+
84
+ ```bash
85
+ scholarfocus --version
86
+ scholarfocus --init-config # writes ~/.config/scholarlib/config.yaml
87
+ ```
88
+
89
+ Everything in that file is optional, but **get the free OpenAlex API key** at
90
+ <https://openalex.org/settings/api> — it raises the daily budget roughly tenfold and
91
+ takes about thirty seconds. Fill in an email address too; it puts you in the polite
92
+ pool at several APIs.
93
+
94
+ Faster fuzzy matching is worth having on large corpora: `uv tool install 'scholarlib[fast]'`.
95
+
96
+ ### 2. The skills
97
+
98
+ **Claude Code**, as a plugin:
99
+
100
+ ```
101
+ /plugin marketplace add wilenius/scholarfocus-skill
102
+ /plugin install scholarlib@wilenius-skills
103
+ ```
104
+
105
+ **Anything else** — copy the skill directories wherever your assistant looks for
106
+ them:
107
+
108
+ ```bash
109
+ git clone https://github.com/wilenius/scholarfocus-skill
110
+ cp -r scholarfocus-skill/skills/* ~/.claude/skills/
111
+ ```
112
+
113
+ The skills call `scholarfocus` and `litreview` as ordinary commands, so they work
114
+ from any directory and need nothing else on the path.
115
+
116
+ ## Usage
117
+
118
+ ```bash
119
+ scholarfocus --researchers "Jane Doe"
120
+ scholarfocus --researchers "0000-0002-1234-5678" # ORCID is always better
121
+ litreview --query "multispecies ethnography" --no-jstor --dry-run
122
+ ```
123
+
124
+ `--dry-run` prints the credit plan and spends nothing. Use it first on an
125
+ unfamiliar topic.
126
+
127
+ ### What is deterministic
128
+
129
+ The library makes no LLM calls. Retrieval, ranking, clustering and rendering are
130
+ plain code, and the output depends only on the API responses. The review prose is
131
+ written by the agent running the skill, from that output.
132
+
133
+ litreview's **thematic strands** (`--cluster`) are counts, not semantics:
134
+
135
+ - `topics` (narrative default): the most frequent OpenAlex topics across the corpus,
136
+ up to 7 with at least 3 works each, become strands. Each work joins the strand of
137
+ its highest-ranked topic. Labels are OpenAlex's topic names, used verbatim.
138
+ - `cocitation`: the references the corpus cites most become anchors. Each strand is
139
+ the works citing one anchor, labelled with the anchor's title.
140
+
141
+ OpenAlex assigns topics with its own classifier, so strands inherit its taxonomy
142
+ and its errors. Treat them as a skeleton to rename, merge or split, not as findings.
143
+
144
+ ## Zotero — recommended
145
+
146
+ **Off by default, and worth turning on.** litreview can read your Zotero library
147
+ over its local API, which changes the output in ways nothing else can:
148
+
149
+ - Results you already own are marked, so a reading list separates what you have
150
+ from what you need to find.
151
+ - Your Better BibTeX citekeys are reused verbatim in exported bibliographies, so
152
+ they match the keys already in your documents.
153
+ - A collection can seed a review through free OpenAlex lookups — a high-precision
154
+ starting set at zero credit cost.
155
+
156
+ Install [Zotero](https://www.zotero.org/download/), then enable *Settings →
157
+ Advanced → "Allow other applications on this computer to communicate with
158
+ Zotero"*. Set `zotero.enabled: true` in your config and pass `--zotero mark` (or
159
+ `both`).
160
+
161
+ For conceptual search over your library — "find things I own that are *about*
162
+ this" — add [`zotero-mcp`](https://github.com/54yyyu/zotero-mcp), an MCP server
163
+ with a vector index over your library. It complements this integration rather
164
+ than replacing it: `zotero-mcp` answers semantic questions, while `--zotero mark`
165
+ does exact ownership matching. Feed DOIs from the former back in via `--seed-doi`.
166
+
167
+ The integration degrades quietly: if Zotero is closed, litreview logs one warning
168
+ and carries on.
169
+
170
+ ## JSTOR — optional
171
+
172
+ litreview can check a local index of JSTOR metadata for books and chapters that
173
+ citation databases miss. It is the difference between a review that sees
174
+ monographs and one that does not, which matters most in the humanities and
175
+ qualitative social sciences.
176
+
177
+ It requires a metadata dump that **JSTOR distributes only to subscribing
178
+ institutions**, at <https://www.jstor.org/ta-support/metadata>. With an
179
+ institutional account you get a gzipped JSONL file of about 1.3 GB, which builds
180
+ into a ~6.7 GB SQLite FTS5 index in roughly eight minutes:
181
+
182
+ ```bash
183
+ # point jstor.source_path at the downloaded file, then
184
+ jstor-index build --all-disciplines # or --disciplines Anthropology,History
185
+ ```
186
+
187
+ Without it, pass `--no-jstor` and litreview will say what the gap costs. See
188
+ [references/JSTOR.md](skills/litreview/references/JSTOR.md) for the join rate and
189
+ what the index does and does not contain.
190
+
191
+ ## Data that does not live in this repo
192
+
193
+ The JSTOR dump belongs outside the working tree — `~/data/jstor/` by default,
194
+ configurable at `jstor.source_path`. The built index lands in
195
+ `~/.local/share/scholarlib/`, the HTTP cache in `~/.cache/scholarlib/`. All are
196
+ gitignored.
197
+
198
+ ## Development
199
+
200
+ ```bash
201
+ git clone https://github.com/wilenius/scholarfocus-skill
202
+ cd scholarfocus-skill
203
+ python -m venv .venv && .venv/bin/pip install -e '.[dev]'
204
+ .venv/bin/python -m pytest
205
+ ```
206
+
207
+ A `config.yaml` at the repo root takes precedence over the user-level one, and is
208
+ gitignored. See [AGENTS.md](AGENTS.md) for versioning, release and packaging
209
+ conventions.
210
+
211
+ ## API status
212
+
213
+ See [docs/api-status.md](docs/api-status.md) for live-verified credential status and
214
+ the measured OpenAlex credit costs that the budget logic is built around.
215
+
216
+ ## Layout
217
+
218
+ ```
219
+ scholarlib/ shared library, published to PyPI
220
+ apis/ one client per data source
221
+ http/ BaseClient, cache, credit ledger
222
+ pipeline/ retrieval and analysis stages
223
+ render/ markdown / json / bibtex output
224
+ jstor/ FTS5 index build and query
225
+ cli/ entry points
226
+ skills/ the two SKILL.md entry points
227
+ tests/ fixtures and unit tests
228
+ ```
229
+
230
+ ## License
231
+
232
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,201 @@
1
+ # scholarfocus-skill
2
+
3
+ Two agent skills over one shared bibliographic library, for Claude Code, OpenClaw
4
+ and any other assistant that can run a shell command.
5
+
6
+ | Skill | Object | Question it answers |
7
+ |---|---|---|
8
+ | [`skills/scholarfocus`](skills/scholarfocus/SKILL.md) | **people** | What does this researcher work on? Who with? What do they cite? |
9
+ | [`skills/litreview`](skills/litreview/SKILL.md) | **topics** | What is the state of the field on X? What are the key works? |
10
+
11
+ Both sit on [`scholarlib`](https://pypi.org/project/scholarlib/), which wraps OpenAlex,
12
+ ORCID, CrossRef, CORE, Unpaywall, OpenAIRE, Semantic Scholar and an optional local
13
+ JSTOR metadata index behind one cache, one credit budget and one deduplication layer.
14
+
15
+ ## Install
16
+
17
+ Two steps: the command-line tool, then the skills.
18
+
19
+ ### 1. The tool
20
+
21
+ Install `scholarlib` as a standalone tool, not into your system Python — on Arch,
22
+ Manjaro, Debian, Ubuntu and Fedora the system interpreter is marked
23
+ externally-managed and a bare `pip install` will refuse with *"This environment is
24
+ externally managed"*. That refusal is correct; do not override it.
25
+
26
+ **macOS**
27
+
28
+ ```bash
29
+ brew install uv # or: brew install pipx && pipx ensurepath
30
+ uv tool install scholarlib
31
+ ```
32
+
33
+ **Linux**
34
+
35
+ ```bash
36
+ # Arch / Manjaro
37
+ sudo pacman -S uv # or: sudo pacman -S python-pipx
38
+
39
+ # Debian / Ubuntu
40
+ sudo apt install pipx # then use `pipx install` below
41
+
42
+ # Fedora
43
+ sudo dnf install uv
44
+
45
+ uv tool install scholarlib # or: pipx install scholarlib
46
+ ```
47
+
48
+ No `uv` or `pipx` available? `python3 -m venv ~/.venvs/scholarlib && ~/.venvs/scholarlib/bin/pip install scholarlib`
49
+ works anywhere, and the commands land in `~/.venvs/scholarlib/bin/`.
50
+
51
+ Verify, and create a config:
52
+
53
+ ```bash
54
+ scholarfocus --version
55
+ scholarfocus --init-config # writes ~/.config/scholarlib/config.yaml
56
+ ```
57
+
58
+ Everything in that file is optional, but **get the free OpenAlex API key** at
59
+ <https://openalex.org/settings/api> — it raises the daily budget roughly tenfold and
60
+ takes about thirty seconds. Fill in an email address too; it puts you in the polite
61
+ pool at several APIs.
62
+
63
+ Faster fuzzy matching is worth having on large corpora: `uv tool install 'scholarlib[fast]'`.
64
+
65
+ ### 2. The skills
66
+
67
+ **Claude Code**, as a plugin:
68
+
69
+ ```
70
+ /plugin marketplace add wilenius/scholarfocus-skill
71
+ /plugin install scholarlib@wilenius-skills
72
+ ```
73
+
74
+ **Anything else** — copy the skill directories wherever your assistant looks for
75
+ them:
76
+
77
+ ```bash
78
+ git clone https://github.com/wilenius/scholarfocus-skill
79
+ cp -r scholarfocus-skill/skills/* ~/.claude/skills/
80
+ ```
81
+
82
+ The skills call `scholarfocus` and `litreview` as ordinary commands, so they work
83
+ from any directory and need nothing else on the path.
84
+
85
+ ## Usage
86
+
87
+ ```bash
88
+ scholarfocus --researchers "Jane Doe"
89
+ scholarfocus --researchers "0000-0002-1234-5678" # ORCID is always better
90
+ litreview --query "multispecies ethnography" --no-jstor --dry-run
91
+ ```
92
+
93
+ `--dry-run` prints the credit plan and spends nothing. Use it first on an
94
+ unfamiliar topic.
95
+
96
+ ### What is deterministic
97
+
98
+ The library makes no LLM calls. Retrieval, ranking, clustering and rendering are
99
+ plain code, and the output depends only on the API responses. The review prose is
100
+ written by the agent running the skill, from that output.
101
+
102
+ litreview's **thematic strands** (`--cluster`) are counts, not semantics:
103
+
104
+ - `topics` (narrative default): the most frequent OpenAlex topics across the corpus,
105
+ up to 7 with at least 3 works each, become strands. Each work joins the strand of
106
+ its highest-ranked topic. Labels are OpenAlex's topic names, used verbatim.
107
+ - `cocitation`: the references the corpus cites most become anchors. Each strand is
108
+ the works citing one anchor, labelled with the anchor's title.
109
+
110
+ OpenAlex assigns topics with its own classifier, so strands inherit its taxonomy
111
+ and its errors. Treat them as a skeleton to rename, merge or split, not as findings.
112
+
113
+ ## Zotero — recommended
114
+
115
+ **Off by default, and worth turning on.** litreview can read your Zotero library
116
+ over its local API, which changes the output in ways nothing else can:
117
+
118
+ - Results you already own are marked, so a reading list separates what you have
119
+ from what you need to find.
120
+ - Your Better BibTeX citekeys are reused verbatim in exported bibliographies, so
121
+ they match the keys already in your documents.
122
+ - A collection can seed a review through free OpenAlex lookups — a high-precision
123
+ starting set at zero credit cost.
124
+
125
+ Install [Zotero](https://www.zotero.org/download/), then enable *Settings →
126
+ Advanced → "Allow other applications on this computer to communicate with
127
+ Zotero"*. Set `zotero.enabled: true` in your config and pass `--zotero mark` (or
128
+ `both`).
129
+
130
+ For conceptual search over your library — "find things I own that are *about*
131
+ this" — add [`zotero-mcp`](https://github.com/54yyyu/zotero-mcp), an MCP server
132
+ with a vector index over your library. It complements this integration rather
133
+ than replacing it: `zotero-mcp` answers semantic questions, while `--zotero mark`
134
+ does exact ownership matching. Feed DOIs from the former back in via `--seed-doi`.
135
+
136
+ The integration degrades quietly: if Zotero is closed, litreview logs one warning
137
+ and carries on.
138
+
139
+ ## JSTOR — optional
140
+
141
+ litreview can check a local index of JSTOR metadata for books and chapters that
142
+ citation databases miss. It is the difference between a review that sees
143
+ monographs and one that does not, which matters most in the humanities and
144
+ qualitative social sciences.
145
+
146
+ It requires a metadata dump that **JSTOR distributes only to subscribing
147
+ institutions**, at <https://www.jstor.org/ta-support/metadata>. With an
148
+ institutional account you get a gzipped JSONL file of about 1.3 GB, which builds
149
+ into a ~6.7 GB SQLite FTS5 index in roughly eight minutes:
150
+
151
+ ```bash
152
+ # point jstor.source_path at the downloaded file, then
153
+ jstor-index build --all-disciplines # or --disciplines Anthropology,History
154
+ ```
155
+
156
+ Without it, pass `--no-jstor` and litreview will say what the gap costs. See
157
+ [references/JSTOR.md](skills/litreview/references/JSTOR.md) for the join rate and
158
+ what the index does and does not contain.
159
+
160
+ ## Data that does not live in this repo
161
+
162
+ The JSTOR dump belongs outside the working tree — `~/data/jstor/` by default,
163
+ configurable at `jstor.source_path`. The built index lands in
164
+ `~/.local/share/scholarlib/`, the HTTP cache in `~/.cache/scholarlib/`. All are
165
+ gitignored.
166
+
167
+ ## Development
168
+
169
+ ```bash
170
+ git clone https://github.com/wilenius/scholarfocus-skill
171
+ cd scholarfocus-skill
172
+ python -m venv .venv && .venv/bin/pip install -e '.[dev]'
173
+ .venv/bin/python -m pytest
174
+ ```
175
+
176
+ A `config.yaml` at the repo root takes precedence over the user-level one, and is
177
+ gitignored. See [AGENTS.md](AGENTS.md) for versioning, release and packaging
178
+ conventions.
179
+
180
+ ## API status
181
+
182
+ See [docs/api-status.md](docs/api-status.md) for live-verified credential status and
183
+ the measured OpenAlex credit costs that the budget logic is built around.
184
+
185
+ ## Layout
186
+
187
+ ```
188
+ scholarlib/ shared library, published to PyPI
189
+ apis/ one client per data source
190
+ http/ BaseClient, cache, credit ledger
191
+ pipeline/ retrieval and analysis stages
192
+ render/ markdown / json / bibtex output
193
+ jstor/ FTS5 index build and query
194
+ cli/ entry points
195
+ skills/ the two SKILL.md entry points
196
+ tests/ fixtures and unit tests
197
+ ```
198
+
199
+ ## License
200
+
201
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,57 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "scholarlib"
7
+ dynamic = ["version"]
8
+ description = "Shared bibliographic-API library behind the scholarfocus and litreview agent skills"
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.11"
13
+ authors = [{ name = "Heikki Wilenius" }]
14
+ keywords = ["bibliography", "openalex", "orcid", "literature-review", "agent-skills"]
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Environment :: Console",
18
+ "Intended Audience :: Science/Research",
19
+ "Programming Language :: Python :: 3.11",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Programming Language :: Python :: 3.13",
22
+ "Topic :: Scientific/Engineering :: Information Analysis",
23
+ "Topic :: Text Processing :: Indexing",
24
+ ]
25
+ dependencies = [
26
+ "requests>=2.31.0",
27
+ "pyyaml>=6.0",
28
+ ]
29
+
30
+ [project.optional-dependencies]
31
+ fast = ["rapidfuzz>=3.6"]
32
+ zotero = ["pyzotero>=1.5"]
33
+ dev = ["pytest>=8.0"]
34
+
35
+ [project.urls]
36
+ Homepage = "https://github.com/wilenius/scholarfocus-skill"
37
+ Repository = "https://github.com/wilenius/scholarfocus-skill"
38
+ Issues = "https://github.com/wilenius/scholarfocus-skill/issues"
39
+
40
+ [project.scripts]
41
+ scholarfocus = "scholarlib.cli.scholarfocus:main"
42
+ litreview = "scholarlib.cli.litreview:main"
43
+ jstor-index = "scholarlib.cli.jstor_index:main"
44
+
45
+ [tool.setuptools.dynamic]
46
+ version = { attr = "scholarlib.__version__" }
47
+
48
+ [tool.setuptools.packages.find]
49
+ include = ["scholarlib*"]
50
+
51
+ [tool.setuptools.package-data]
52
+ scholarlib = ["config.example.yaml", "jstor/*.sql"]
53
+
54
+ [tool.pyright]
55
+ pythonVersion = "3.11"
56
+ include = ["scholarlib", "tests"]
57
+ reportMissingImports = "warning"
@@ -0,0 +1,3 @@
1
+ """Shared bibliographic-API library behind the scholarfocus and litreview skills."""
2
+
3
+ __version__ = "0.3.0"
@@ -0,0 +1,94 @@
1
+ """Convert each source's native shape into the canonical Record."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Optional
6
+
7
+ from scholarlib.apis.openalex import reconstruct_abstract
8
+ from scholarlib.records import Record
9
+
10
+
11
+ def _authors_from_openalex(work: dict) -> list[str]:
12
+ out = []
13
+ for a in work.get("authorships") or []:
14
+ name = ((a.get("author") or {}).get("display_name"))
15
+ if name:
16
+ out.append(name)
17
+ return out
18
+
19
+
20
+ def _topics_from_openalex(work: dict) -> list[str]:
21
+ names = []
22
+ for t in (work.get("topics") or [])[:5]:
23
+ if t.get("display_name"):
24
+ names.append(t["display_name"])
25
+ for c in (work.get("concepts") or [])[:5]:
26
+ if c.get("display_name") and c["display_name"] not in names:
27
+ names.append(c["display_name"])
28
+ return names
29
+
30
+
31
+ def from_openalex_work(work: dict, *, provenance: Optional[str] = None) -> Record:
32
+ abstract = reconstruct_abstract(work.get("abstract_inverted_index"))
33
+ loc = work.get("primary_location") or {}
34
+ src = loc.get("source") or {}
35
+ best_oa = work.get("best_oa_location") or {}
36
+ oa = work.get("open_access") or {}
37
+ return Record(
38
+ doi=work.get("doi"),
39
+ openalex_id=work.get("id"),
40
+ title=work.get("title") or work.get("display_name"),
41
+ authors=_authors_from_openalex(work),
42
+ year=work.get("publication_year"),
43
+ venue=src.get("display_name"),
44
+ type=work.get("type"),
45
+ language=work.get("language"),
46
+ abstract=abstract,
47
+ abstract_source="openalex" if abstract else None,
48
+ cited_by_count=work.get("cited_by_count"),
49
+ referenced_works=list(work.get("referenced_works") or []),
50
+ related_works=list(work.get("related_works") or []),
51
+ topics=_topics_from_openalex(work),
52
+ oa_status=oa.get("oa_status"),
53
+ oa_url=best_oa.get("pdf_url") or best_oa.get("landing_page_url"),
54
+ provenance=[provenance] if provenance else [],
55
+ )
56
+
57
+
58
+ def from_crossref_item(item: dict, *, provenance: str = "crossref") -> Record:
59
+ titles = item.get("title") or []
60
+ containers = item.get("container-title") or []
61
+ authors = []
62
+ for a in item.get("author") or []:
63
+ nm = " ".join(x for x in (a.get("given"), a.get("family")) if x) or a.get("name")
64
+ if nm:
65
+ authors.append(nm)
66
+ issued = ((item.get("issued") or {}).get("date-parts") or [[None]])[0]
67
+ return Record(
68
+ doi=item.get("DOI"),
69
+ title=titles[0] if titles else None,
70
+ authors=authors,
71
+ year=issued[0] if issued else None,
72
+ venue=containers[0] if containers else None,
73
+ type=item.get("type"),
74
+ cited_by_count=item.get("is-referenced-by-count"),
75
+ provenance=[provenance],
76
+ )
77
+
78
+
79
+ def from_s2_paper(paper: dict, *, provenance: str = "semantic_scholar") -> Record:
80
+ ext = paper.get("externalIds") or {}
81
+ tldr = (paper.get("tldr") or {}).get("text")
82
+ abstract = paper.get("abstract") or tldr
83
+ return Record(
84
+ doi=ext.get("DOI"),
85
+ title=paper.get("title"),
86
+ year=paper.get("year"),
87
+ venue=paper.get("venue"),
88
+ abstract=abstract,
89
+ abstract_source=("semantic_scholar" if paper.get("abstract")
90
+ else ("s2_tldr" if tldr else None)),
91
+ cited_by_count=paper.get("citationCount"),
92
+ topics=list(paper.get("fieldsOfStudy") or []),
93
+ provenance=[provenance],
94
+ )
File without changes
@@ -0,0 +1,41 @@
1
+ """CORE client. Key is optional but raises the rate limit from 10/min to 150/min."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from typing import Optional
7
+
8
+ from scholarlib.http.base import BaseClient
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+ BASE_URL = "https://api.core.ac.uk/v3"
13
+
14
+
15
+ class COREClient(BaseClient):
16
+ name = "core"
17
+ base_url = BASE_URL
18
+
19
+ def __init__(self, api_key: Optional[str] = None, **kw):
20
+ super().__init__(**kw)
21
+ self.api_key = api_key
22
+ if not api_key:
23
+ logger.debug("CORE: no API key — rate limit is 10/min instead of 150/min")
24
+
25
+ def _auth_headers(self) -> dict:
26
+ return {"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}
27
+
28
+ def search_works(self, query: str, limit: int = 10) -> list[dict]:
29
+ data = self.get("/search/works", {"q": query, "limit": limit})
30
+ return (data or {}).get("results", [])
31
+
32
+ def get_work_by_doi(self, doi: str) -> Optional[dict]:
33
+ d = str(doi).replace("https://doi.org/", "").strip()
34
+ results = self.search_works(f'doi:"{d}"', limit=1)
35
+ return results[0] if results else None
36
+
37
+ def get_abstract(self, doi: str) -> Optional[str]:
38
+ w = self.get_work_by_doi(doi)
39
+ if not w:
40
+ return None
41
+ return w.get("abstract") or w.get("description") or None