scholarlib 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scholarlib-0.3.0/LICENSE +21 -0
- scholarlib-0.3.0/PKG-INFO +232 -0
- scholarlib-0.3.0/README.md +201 -0
- scholarlib-0.3.0/pyproject.toml +57 -0
- scholarlib-0.3.0/scholarlib/__init__.py +3 -0
- scholarlib-0.3.0/scholarlib/adapters.py +94 -0
- scholarlib-0.3.0/scholarlib/apis/__init__.py +0 -0
- scholarlib-0.3.0/scholarlib/apis/core_api.py +41 -0
- scholarlib-0.3.0/scholarlib/apis/crossref.py +93 -0
- scholarlib-0.3.0/scholarlib/apis/openaire.py +88 -0
- scholarlib-0.3.0/scholarlib/apis/openalex.py +199 -0
- scholarlib-0.3.0/scholarlib/apis/orcid.py +118 -0
- scholarlib-0.3.0/scholarlib/apis/semantic_scholar.py +99 -0
- scholarlib-0.3.0/scholarlib/apis/unpaywall.py +47 -0
- scholarlib-0.3.0/scholarlib/apis/zotero_local.py +156 -0
- scholarlib-0.3.0/scholarlib/cli/__init__.py +0 -0
- scholarlib-0.3.0/scholarlib/cli/jstor_index.py +165 -0
- scholarlib-0.3.0/scholarlib/cli/litreview.py +326 -0
- scholarlib-0.3.0/scholarlib/cli/scholarfocus.py +662 -0
- scholarlib-0.3.0/scholarlib/config.example.yaml +68 -0
- scholarlib-0.3.0/scholarlib/config.py +172 -0
- scholarlib-0.3.0/scholarlib/dedup.py +177 -0
- scholarlib-0.3.0/scholarlib/http/__init__.py +0 -0
- scholarlib-0.3.0/scholarlib/http/base.py +212 -0
- scholarlib-0.3.0/scholarlib/http/budget.py +162 -0
- scholarlib-0.3.0/scholarlib/http/cache.py +144 -0
- scholarlib-0.3.0/scholarlib/http/policy.py +59 -0
- scholarlib-0.3.0/scholarlib/jstor/__init__.py +0 -0
- scholarlib-0.3.0/scholarlib/jstor/build.py +336 -0
- scholarlib-0.3.0/scholarlib/jstor/indexes.sql +8 -0
- scholarlib-0.3.0/scholarlib/jstor/query.py +295 -0
- scholarlib-0.3.0/scholarlib/jstor/schema.sql +48 -0
- scholarlib-0.3.0/scholarlib/pipeline/__init__.py +0 -0
- scholarlib-0.3.0/scholarlib/pipeline/abstracts.py +59 -0
- scholarlib-0.3.0/scholarlib/pipeline/cluster.py +117 -0
- scholarlib-0.3.0/scholarlib/pipeline/context.py +92 -0
- scholarlib-0.3.0/scholarlib/pipeline/disambiguate.py +377 -0
- scholarlib-0.3.0/scholarlib/pipeline/jstor_coverage.py +137 -0
- scholarlib-0.3.0/scholarlib/pipeline/oa_links.py +61 -0
- scholarlib-0.3.0/scholarlib/pipeline/profile.py +333 -0
- scholarlib-0.3.0/scholarlib/pipeline/rank.py +139 -0
- scholarlib-0.3.0/scholarlib/pipeline/seed.py +97 -0
- scholarlib-0.3.0/scholarlib/pipeline/snowball.py +120 -0
- scholarlib-0.3.0/scholarlib/progress.py +131 -0
- scholarlib-0.3.0/scholarlib/records.py +202 -0
- scholarlib-0.3.0/scholarlib/render/__init__.py +0 -0
- scholarlib-0.3.0/scholarlib/render/bib.py +139 -0
- scholarlib-0.3.0/scholarlib/render/json_out.py +120 -0
- scholarlib-0.3.0/scholarlib/render/narrative.py +167 -0
- scholarlib-0.3.0/scholarlib/render/systematic.py +93 -0
- scholarlib-0.3.0/scholarlib.egg-info/PKG-INFO +232 -0
- scholarlib-0.3.0/scholarlib.egg-info/SOURCES.txt +59 -0
- scholarlib-0.3.0/scholarlib.egg-info/dependency_links.txt +1 -0
- scholarlib-0.3.0/scholarlib.egg-info/entry_points.txt +4 -0
- scholarlib-0.3.0/scholarlib.egg-info/requires.txt +11 -0
- scholarlib-0.3.0/scholarlib.egg-info/top_level.txt +1 -0
- scholarlib-0.3.0/setup.cfg +4 -0
- scholarlib-0.3.0/tests/test_budget.py +46 -0
- scholarlib-0.3.0/tests/test_dedup.py +155 -0
- scholarlib-0.3.0/tests/test_disambiguate.py +90 -0
- scholarlib-0.3.0/tests/test_jstor_query.py +29 -0
scholarlib-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Heikki Wilenius
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scholarlib
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Shared bibliographic-API library behind the scholarfocus and litreview agent skills
|
|
5
|
+
Author: Heikki Wilenius
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/wilenius/scholarfocus-skill
|
|
8
|
+
Project-URL: Repository, https://github.com/wilenius/scholarfocus-skill
|
|
9
|
+
Project-URL: Issues, https://github.com/wilenius/scholarfocus-skill/issues
|
|
10
|
+
Keywords: bibliography,openalex,orcid,literature-review,agent-skills
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
18
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
19
|
+
Requires-Python: >=3.11
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: requests>=2.31.0
|
|
23
|
+
Requires-Dist: pyyaml>=6.0
|
|
24
|
+
Provides-Extra: fast
|
|
25
|
+
Requires-Dist: rapidfuzz>=3.6; extra == "fast"
|
|
26
|
+
Provides-Extra: zotero
|
|
27
|
+
Requires-Dist: pyzotero>=1.5; extra == "zotero"
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# scholarfocus-skill
|
|
33
|
+
|
|
34
|
+
Two agent skills over one shared bibliographic library, for Claude Code, OpenClaw
|
|
35
|
+
and any other assistant that can run a shell command.
|
|
36
|
+
|
|
37
|
+
| Skill | Object | Question it answers |
|
|
38
|
+
|---|---|---|
|
|
39
|
+
| [`skills/scholarfocus`](skills/scholarfocus/SKILL.md) | **people** | What does this researcher work on? Who with? What do they cite? |
|
|
40
|
+
| [`skills/litreview`](skills/litreview/SKILL.md) | **topics** | What is the state of the field on X? What are the key works? |
|
|
41
|
+
|
|
42
|
+
Both sit on [`scholarlib`](https://pypi.org/project/scholarlib/), which wraps OpenAlex,
|
|
43
|
+
ORCID, CrossRef, CORE, Unpaywall, OpenAIRE, Semantic Scholar and an optional local
|
|
44
|
+
JSTOR metadata index behind one cache, one credit budget and one deduplication layer.
|
|
45
|
+
|
|
46
|
+
## Install
|
|
47
|
+
|
|
48
|
+
Two steps: the command-line tool, then the skills.
|
|
49
|
+
|
|
50
|
+
### 1. The tool
|
|
51
|
+
|
|
52
|
+
Install `scholarlib` as a standalone tool, not into your system Python — on Arch,
|
|
53
|
+
Manjaro, Debian, Ubuntu and Fedora the system interpreter is marked
|
|
54
|
+
externally-managed and a bare `pip install` will refuse with *"This environment is
|
|
55
|
+
externally managed"*. That refusal is correct; do not override it.
|
|
56
|
+
|
|
57
|
+
**macOS**
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
brew install uv # or: brew install pipx && pipx ensurepath
|
|
61
|
+
uv tool install scholarlib
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
**Linux**
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
# Arch / Manjaro
|
|
68
|
+
sudo pacman -S uv # or: sudo pacman -S python-pipx
|
|
69
|
+
|
|
70
|
+
# Debian / Ubuntu
|
|
71
|
+
sudo apt install pipx # then use `pipx install` below
|
|
72
|
+
|
|
73
|
+
# Fedora
|
|
74
|
+
sudo dnf install uv
|
|
75
|
+
|
|
76
|
+
uv tool install scholarlib # or: pipx install scholarlib
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
No `uv` or `pipx` available? `python3 -m venv ~/.venvs/scholarlib && ~/.venvs/scholarlib/bin/pip install scholarlib`
|
|
80
|
+
works anywhere, and the commands land in `~/.venvs/scholarlib/bin/`.
|
|
81
|
+
|
|
82
|
+
Verify, and create a config:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
scholarfocus --version
|
|
86
|
+
scholarfocus --init-config # writes ~/.config/scholarlib/config.yaml
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Everything in that file is optional, but **get the free OpenAlex API key** at
|
|
90
|
+
<https://openalex.org/settings/api> — it raises the daily budget roughly tenfold and
|
|
91
|
+
takes about thirty seconds. Fill in an email address too; it puts you in the polite
|
|
92
|
+
pool at several APIs.
|
|
93
|
+
|
|
94
|
+
Faster fuzzy matching is worth having on large corpora: `uv tool install 'scholarlib[fast]'`.
|
|
95
|
+
|
|
96
|
+
### 2. The skills
|
|
97
|
+
|
|
98
|
+
**Claude Code**, as a plugin:
|
|
99
|
+
|
|
100
|
+
```
|
|
101
|
+
/plugin marketplace add wilenius/scholarfocus-skill
|
|
102
|
+
/plugin install scholarlib@wilenius-skills
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
**Anything else** — copy the skill directories wherever your assistant looks for
|
|
106
|
+
them:
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
git clone https://github.com/wilenius/scholarfocus-skill
|
|
110
|
+
cp -r scholarfocus-skill/skills/* ~/.claude/skills/
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
The skills call `scholarfocus` and `litreview` as ordinary commands, so they work
|
|
114
|
+
from any directory and need nothing else on the path.
|
|
115
|
+
|
|
116
|
+
## Usage
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
scholarfocus --researchers "Jane Doe"
|
|
120
|
+
scholarfocus --researchers "0000-0002-1234-5678" # ORCID is always better
|
|
121
|
+
litreview --query "multispecies ethnography" --no-jstor --dry-run
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`--dry-run` prints the credit plan and spends nothing. Use it first on an
|
|
125
|
+
unfamiliar topic.
|
|
126
|
+
|
|
127
|
+
### What is deterministic
|
|
128
|
+
|
|
129
|
+
The library makes no LLM calls. Retrieval, ranking, clustering and rendering are
|
|
130
|
+
plain code, and the output depends only on the API responses. The review prose is
|
|
131
|
+
written by the agent running the skill, from that output.
|
|
132
|
+
|
|
133
|
+
litreview's **thematic strands** (`--cluster`) are counts, not semantics:
|
|
134
|
+
|
|
135
|
+
- `topics` (narrative default): the most frequent OpenAlex topics across the corpus,
|
|
136
|
+
up to 7 with at least 3 works each, become strands. Each work joins the strand of
|
|
137
|
+
its highest-ranked topic. Labels are OpenAlex's topic names, used verbatim.
|
|
138
|
+
- `cocitation`: the references the corpus cites most become anchors. Each strand is
|
|
139
|
+
the works citing one anchor, labelled with the anchor's title.
|
|
140
|
+
|
|
141
|
+
OpenAlex assigns topics with its own classifier, so strands inherit its taxonomy
|
|
142
|
+
and its errors. Treat them as a skeleton to rename, merge or split, not as findings.
|
|
143
|
+
|
|
144
|
+
## Zotero — recommended
|
|
145
|
+
|
|
146
|
+
**Off by default, and worth turning on.** litreview can read your Zotero library
|
|
147
|
+
over its local API, which changes the output in ways nothing else can:
|
|
148
|
+
|
|
149
|
+
- Results you already own are marked, so a reading list separates what you have
|
|
150
|
+
from what you need to find.
|
|
151
|
+
- Your Better BibTeX citekeys are reused verbatim in exported bibliographies, so
|
|
152
|
+
they match the keys already in your documents.
|
|
153
|
+
- A collection can seed a review through free OpenAlex lookups — a high-precision
|
|
154
|
+
starting set at zero credit cost.
|
|
155
|
+
|
|
156
|
+
Install [Zotero](https://www.zotero.org/download/), then enable *Settings →
|
|
157
|
+
Advanced → "Allow other applications on this computer to communicate with
|
|
158
|
+
Zotero"*. Set `zotero.enabled: true` in your config and pass `--zotero mark` (or
|
|
159
|
+
`both`).
|
|
160
|
+
|
|
161
|
+
For conceptual search over your library — "find things I own that are *about*
|
|
162
|
+
this" — add [`zotero-mcp`](https://github.com/54yyyu/zotero-mcp), an MCP server
|
|
163
|
+
with a vector index over your library. It complements this integration rather
|
|
164
|
+
than replacing it: `zotero-mcp` answers semantic questions, while `--zotero mark`
|
|
165
|
+
does exact ownership matching. Feed DOIs from the former back in via `--seed-doi`.
|
|
166
|
+
|
|
167
|
+
The integration degrades quietly: if Zotero is closed, litreview logs one warning
|
|
168
|
+
and carries on.
|
|
169
|
+
|
|
170
|
+
## JSTOR — optional
|
|
171
|
+
|
|
172
|
+
litreview can check a local index of JSTOR metadata for books and chapters that
|
|
173
|
+
citation databases miss. It is the difference between a review that sees
|
|
174
|
+
monographs and one that does not, which matters most in the humanities and
|
|
175
|
+
qualitative social sciences.
|
|
176
|
+
|
|
177
|
+
It requires a metadata dump that **JSTOR distributes only to subscribing
|
|
178
|
+
institutions**, at <https://www.jstor.org/ta-support/metadata>. With an
|
|
179
|
+
institutional account you get a gzipped JSONL file of about 1.3 GB, which builds
|
|
180
|
+
into a ~6.7 GB SQLite FTS5 index in roughly eight minutes:
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
# point jstor.source_path at the downloaded file, then
|
|
184
|
+
jstor-index build --all-disciplines # or --disciplines Anthropology,History
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
Without it, pass `--no-jstor` and litreview will say what the gap costs. See
|
|
188
|
+
[references/JSTOR.md](skills/litreview/references/JSTOR.md) for the join rate and
|
|
189
|
+
what the index does and does not contain.
|
|
190
|
+
|
|
191
|
+
## Data that does not live in this repo
|
|
192
|
+
|
|
193
|
+
The JSTOR dump belongs outside the working tree — `~/data/jstor/` by default,
|
|
194
|
+
configurable at `jstor.source_path`. The built index lands in
|
|
195
|
+
`~/.local/share/scholarlib/`, the HTTP cache in `~/.cache/scholarlib/`. All are
|
|
196
|
+
gitignored.
|
|
197
|
+
|
|
198
|
+
## Development
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
git clone https://github.com/wilenius/scholarfocus-skill
|
|
202
|
+
cd scholarfocus-skill
|
|
203
|
+
python -m venv .venv && .venv/bin/pip install -e '.[dev]'
|
|
204
|
+
.venv/bin/python -m pytest
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
A `config.yaml` at the repo root takes precedence over the user-level one, and is
|
|
208
|
+
gitignored. See [AGENTS.md](AGENTS.md) for versioning, release and packaging
|
|
209
|
+
conventions.
|
|
210
|
+
|
|
211
|
+
## API status
|
|
212
|
+
|
|
213
|
+
See [docs/api-status.md](docs/api-status.md) for live-verified credential status and
|
|
214
|
+
the measured OpenAlex credit costs that the budget logic is built around.
|
|
215
|
+
|
|
216
|
+
## Layout
|
|
217
|
+
|
|
218
|
+
```
|
|
219
|
+
scholarlib/ shared library, published to PyPI
|
|
220
|
+
apis/ one client per data source
|
|
221
|
+
http/ BaseClient, cache, credit ledger
|
|
222
|
+
pipeline/ retrieval and analysis stages
|
|
223
|
+
render/ markdown / json / bibtex output
|
|
224
|
+
jstor/ FTS5 index build and query
|
|
225
|
+
cli/ entry points
|
|
226
|
+
skills/ the two SKILL.md entry points
|
|
227
|
+
tests/ fixtures and unit tests
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
## License
|
|
231
|
+
|
|
232
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# scholarfocus-skill
|
|
2
|
+
|
|
3
|
+
Two agent skills over one shared bibliographic library, for Claude Code, OpenClaw
|
|
4
|
+
and any other assistant that can run a shell command.
|
|
5
|
+
|
|
6
|
+
| Skill | Object | Question it answers |
|
|
7
|
+
|---|---|---|
|
|
8
|
+
| [`skills/scholarfocus`](skills/scholarfocus/SKILL.md) | **people** | What does this researcher work on? Who with? What do they cite? |
|
|
9
|
+
| [`skills/litreview`](skills/litreview/SKILL.md) | **topics** | What is the state of the field on X? What are the key works? |
|
|
10
|
+
|
|
11
|
+
Both sit on [`scholarlib`](https://pypi.org/project/scholarlib/), which wraps OpenAlex,
|
|
12
|
+
ORCID, CrossRef, CORE, Unpaywall, OpenAIRE, Semantic Scholar and an optional local
|
|
13
|
+
JSTOR metadata index behind one cache, one credit budget and one deduplication layer.
|
|
14
|
+
|
|
15
|
+
## Install
|
|
16
|
+
|
|
17
|
+
Two steps: the command-line tool, then the skills.
|
|
18
|
+
|
|
19
|
+
### 1. The tool
|
|
20
|
+
|
|
21
|
+
Install `scholarlib` as a standalone tool, not into your system Python — on Arch,
|
|
22
|
+
Manjaro, Debian, Ubuntu and Fedora the system interpreter is marked
|
|
23
|
+
externally-managed and a bare `pip install` will refuse with *"This environment is
|
|
24
|
+
externally managed"*. That refusal is correct; do not override it.
|
|
25
|
+
|
|
26
|
+
**macOS**
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
brew install uv # or: brew install pipx && pipx ensurepath
|
|
30
|
+
uv tool install scholarlib
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
**Linux**
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
# Arch / Manjaro
|
|
37
|
+
sudo pacman -S uv # or: sudo pacman -S python-pipx
|
|
38
|
+
|
|
39
|
+
# Debian / Ubuntu
|
|
40
|
+
sudo apt install pipx # then use `pipx install` below
|
|
41
|
+
|
|
42
|
+
# Fedora
|
|
43
|
+
sudo dnf install uv
|
|
44
|
+
|
|
45
|
+
uv tool install scholarlib # or: pipx install scholarlib
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
No `uv` or `pipx` available? `python3 -m venv ~/.venvs/scholarlib && ~/.venvs/scholarlib/bin/pip install scholarlib`
|
|
49
|
+
works anywhere, and the commands land in `~/.venvs/scholarlib/bin/`.
|
|
50
|
+
|
|
51
|
+
Verify, and create a config:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
scholarfocus --version
|
|
55
|
+
scholarfocus --init-config # writes ~/.config/scholarlib/config.yaml
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Everything in that file is optional, but **get the free OpenAlex API key** at
|
|
59
|
+
<https://openalex.org/settings/api> — it raises the daily budget roughly tenfold and
|
|
60
|
+
takes about thirty seconds. Fill in an email address too; it puts you in the polite
|
|
61
|
+
pool at several APIs.
|
|
62
|
+
|
|
63
|
+
Faster fuzzy matching is worth having on large corpora: `uv tool install 'scholarlib[fast]'`.
|
|
64
|
+
|
|
65
|
+
### 2. The skills
|
|
66
|
+
|
|
67
|
+
**Claude Code**, as a plugin:
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
/plugin marketplace add wilenius/scholarfocus-skill
|
|
71
|
+
/plugin install scholarlib@wilenius-skills
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
**Anything else** — copy the skill directories wherever your assistant looks for
|
|
75
|
+
them:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
git clone https://github.com/wilenius/scholarfocus-skill
|
|
79
|
+
cp -r scholarfocus-skill/skills/* ~/.claude/skills/
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
The skills call `scholarfocus` and `litreview` as ordinary commands, so they work
|
|
83
|
+
from any directory and need nothing else on the path.
|
|
84
|
+
|
|
85
|
+
## Usage
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
scholarfocus --researchers "Jane Doe"
|
|
89
|
+
scholarfocus --researchers "0000-0002-1234-5678" # ORCID is always better
|
|
90
|
+
litreview --query "multispecies ethnography" --no-jstor --dry-run
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
`--dry-run` prints the credit plan and spends nothing. Use it first on an
|
|
94
|
+
unfamiliar topic.
|
|
95
|
+
|
|
96
|
+
### What is deterministic
|
|
97
|
+
|
|
98
|
+
The library makes no LLM calls. Retrieval, ranking, clustering and rendering are
|
|
99
|
+
plain code, and the output depends only on the API responses. The review prose is
|
|
100
|
+
written by the agent running the skill, from that output.
|
|
101
|
+
|
|
102
|
+
litreview's **thematic strands** (`--cluster`) are counts, not semantics:
|
|
103
|
+
|
|
104
|
+
- `topics` (narrative default): the most frequent OpenAlex topics across the corpus,
|
|
105
|
+
up to 7 with at least 3 works each, become strands. Each work joins the strand of
|
|
106
|
+
its highest-ranked topic. Labels are OpenAlex's topic names, used verbatim.
|
|
107
|
+
- `cocitation`: the references the corpus cites most become anchors. Each strand is
|
|
108
|
+
the works citing one anchor, labelled with the anchor's title.
|
|
109
|
+
|
|
110
|
+
OpenAlex assigns topics with its own classifier, so strands inherit its taxonomy
|
|
111
|
+
and its errors. Treat them as a skeleton to rename, merge or split, not as findings.
|
|
112
|
+
|
|
113
|
+
## Zotero — recommended
|
|
114
|
+
|
|
115
|
+
**Off by default, and worth turning on.** litreview can read your Zotero library
|
|
116
|
+
over its local API, which changes the output in ways nothing else can:
|
|
117
|
+
|
|
118
|
+
- Results you already own are marked, so a reading list separates what you have
|
|
119
|
+
from what you need to find.
|
|
120
|
+
- Your Better BibTeX citekeys are reused verbatim in exported bibliographies, so
|
|
121
|
+
they match the keys already in your documents.
|
|
122
|
+
- A collection can seed a review through free OpenAlex lookups — a high-precision
|
|
123
|
+
starting set at zero credit cost.
|
|
124
|
+
|
|
125
|
+
Install [Zotero](https://www.zotero.org/download/), then enable *Settings →
|
|
126
|
+
Advanced → "Allow other applications on this computer to communicate with
|
|
127
|
+
Zotero"*. Set `zotero.enabled: true` in your config and pass `--zotero mark` (or
|
|
128
|
+
`both`).
|
|
129
|
+
|
|
130
|
+
For conceptual search over your library — "find things I own that are *about*
|
|
131
|
+
this" — add [`zotero-mcp`](https://github.com/54yyyu/zotero-mcp), an MCP server
|
|
132
|
+
with a vector index over your library. It complements this integration rather
|
|
133
|
+
than replacing it: `zotero-mcp` answers semantic questions, while `--zotero mark`
|
|
134
|
+
does exact ownership matching. Feed DOIs from the former back in via `--seed-doi`.
|
|
135
|
+
|
|
136
|
+
The integration degrades quietly: if Zotero is closed, litreview logs one warning
|
|
137
|
+
and carries on.
|
|
138
|
+
|
|
139
|
+
## JSTOR — optional
|
|
140
|
+
|
|
141
|
+
litreview can check a local index of JSTOR metadata for books and chapters that
|
|
142
|
+
citation databases miss. It is the difference between a review that sees
|
|
143
|
+
monographs and one that does not, which matters most in the humanities and
|
|
144
|
+
qualitative social sciences.
|
|
145
|
+
|
|
146
|
+
It requires a metadata dump that **JSTOR distributes only to subscribing
|
|
147
|
+
institutions**, at <https://www.jstor.org/ta-support/metadata>. With an
|
|
148
|
+
institutional account you get a gzipped JSONL file of about 1.3 GB, which builds
|
|
149
|
+
into a ~6.7 GB SQLite FTS5 index in roughly eight minutes:
|
|
150
|
+
|
|
151
|
+
```bash
|
|
152
|
+
# point jstor.source_path at the downloaded file, then
|
|
153
|
+
jstor-index build --all-disciplines # or --disciplines Anthropology,History
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Without it, pass `--no-jstor` and litreview will say what the gap costs. See
|
|
157
|
+
[references/JSTOR.md](skills/litreview/references/JSTOR.md) for the join rate and
|
|
158
|
+
what the index does and does not contain.
|
|
159
|
+
|
|
160
|
+
## Data that does not live in this repo
|
|
161
|
+
|
|
162
|
+
The JSTOR dump belongs outside the working tree — `~/data/jstor/` by default,
|
|
163
|
+
configurable at `jstor.source_path`. The built index lands in
|
|
164
|
+
`~/.local/share/scholarlib/`, the HTTP cache in `~/.cache/scholarlib/`. All are
|
|
165
|
+
gitignored.
|
|
166
|
+
|
|
167
|
+
## Development
|
|
168
|
+
|
|
169
|
+
```bash
|
|
170
|
+
git clone https://github.com/wilenius/scholarfocus-skill
|
|
171
|
+
cd scholarfocus-skill
|
|
172
|
+
python -m venv .venv && .venv/bin/pip install -e '.[dev]'
|
|
173
|
+
.venv/bin/python -m pytest
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
A `config.yaml` at the repo root takes precedence over the user-level one, and is
|
|
177
|
+
gitignored. See [AGENTS.md](AGENTS.md) for versioning, release and packaging
|
|
178
|
+
conventions.
|
|
179
|
+
|
|
180
|
+
## API status
|
|
181
|
+
|
|
182
|
+
See [docs/api-status.md](docs/api-status.md) for live-verified credential status and
|
|
183
|
+
the measured OpenAlex credit costs that the budget logic is built around.
|
|
184
|
+
|
|
185
|
+
## Layout
|
|
186
|
+
|
|
187
|
+
```
|
|
188
|
+
scholarlib/ shared library, published to PyPI
|
|
189
|
+
apis/ one client per data source
|
|
190
|
+
http/ BaseClient, cache, credit ledger
|
|
191
|
+
pipeline/ retrieval and analysis stages
|
|
192
|
+
render/ markdown / json / bibtex output
|
|
193
|
+
jstor/ FTS5 index build and query
|
|
194
|
+
cli/ entry points
|
|
195
|
+
skills/ the two SKILL.md entry points
|
|
196
|
+
tests/ fixtures and unit tests
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
## License
|
|
200
|
+
|
|
201
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "scholarlib"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Shared bibliographic-API library behind the scholarfocus and litreview agent skills"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.11"
|
|
13
|
+
authors = [{ name = "Heikki Wilenius" }]
|
|
14
|
+
keywords = ["bibliography", "openalex", "orcid", "literature-review", "agent-skills"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Environment :: Console",
|
|
18
|
+
"Intended Audience :: Science/Research",
|
|
19
|
+
"Programming Language :: Python :: 3.11",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Programming Language :: Python :: 3.13",
|
|
22
|
+
"Topic :: Scientific/Engineering :: Information Analysis",
|
|
23
|
+
"Topic :: Text Processing :: Indexing",
|
|
24
|
+
]
|
|
25
|
+
dependencies = [
|
|
26
|
+
"requests>=2.31.0",
|
|
27
|
+
"pyyaml>=6.0",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[project.optional-dependencies]
|
|
31
|
+
fast = ["rapidfuzz>=3.6"]
|
|
32
|
+
zotero = ["pyzotero>=1.5"]
|
|
33
|
+
dev = ["pytest>=8.0"]
|
|
34
|
+
|
|
35
|
+
[project.urls]
|
|
36
|
+
Homepage = "https://github.com/wilenius/scholarfocus-skill"
|
|
37
|
+
Repository = "https://github.com/wilenius/scholarfocus-skill"
|
|
38
|
+
Issues = "https://github.com/wilenius/scholarfocus-skill/issues"
|
|
39
|
+
|
|
40
|
+
[project.scripts]
|
|
41
|
+
scholarfocus = "scholarlib.cli.scholarfocus:main"
|
|
42
|
+
litreview = "scholarlib.cli.litreview:main"
|
|
43
|
+
jstor-index = "scholarlib.cli.jstor_index:main"
|
|
44
|
+
|
|
45
|
+
[tool.setuptools.dynamic]
|
|
46
|
+
version = { attr = "scholarlib.__version__" }
|
|
47
|
+
|
|
48
|
+
[tool.setuptools.packages.find]
|
|
49
|
+
include = ["scholarlib*"]
|
|
50
|
+
|
|
51
|
+
[tool.setuptools.package-data]
|
|
52
|
+
scholarlib = ["config.example.yaml", "jstor/*.sql"]
|
|
53
|
+
|
|
54
|
+
[tool.pyright]
|
|
55
|
+
pythonVersion = "3.11"
|
|
56
|
+
include = ["scholarlib", "tests"]
|
|
57
|
+
reportMissingImports = "warning"
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Convert each source's native shape into the canonical Record."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from scholarlib.apis.openalex import reconstruct_abstract
|
|
8
|
+
from scholarlib.records import Record
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _authors_from_openalex(work: dict) -> list[str]:
|
|
12
|
+
out = []
|
|
13
|
+
for a in work.get("authorships") or []:
|
|
14
|
+
name = ((a.get("author") or {}).get("display_name"))
|
|
15
|
+
if name:
|
|
16
|
+
out.append(name)
|
|
17
|
+
return out
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _topics_from_openalex(work: dict) -> list[str]:
|
|
21
|
+
names = []
|
|
22
|
+
for t in (work.get("topics") or [])[:5]:
|
|
23
|
+
if t.get("display_name"):
|
|
24
|
+
names.append(t["display_name"])
|
|
25
|
+
for c in (work.get("concepts") or [])[:5]:
|
|
26
|
+
if c.get("display_name") and c["display_name"] not in names:
|
|
27
|
+
names.append(c["display_name"])
|
|
28
|
+
return names
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def from_openalex_work(work: dict, *, provenance: Optional[str] = None) -> Record:
|
|
32
|
+
abstract = reconstruct_abstract(work.get("abstract_inverted_index"))
|
|
33
|
+
loc = work.get("primary_location") or {}
|
|
34
|
+
src = loc.get("source") or {}
|
|
35
|
+
best_oa = work.get("best_oa_location") or {}
|
|
36
|
+
oa = work.get("open_access") or {}
|
|
37
|
+
return Record(
|
|
38
|
+
doi=work.get("doi"),
|
|
39
|
+
openalex_id=work.get("id"),
|
|
40
|
+
title=work.get("title") or work.get("display_name"),
|
|
41
|
+
authors=_authors_from_openalex(work),
|
|
42
|
+
year=work.get("publication_year"),
|
|
43
|
+
venue=src.get("display_name"),
|
|
44
|
+
type=work.get("type"),
|
|
45
|
+
language=work.get("language"),
|
|
46
|
+
abstract=abstract,
|
|
47
|
+
abstract_source="openalex" if abstract else None,
|
|
48
|
+
cited_by_count=work.get("cited_by_count"),
|
|
49
|
+
referenced_works=list(work.get("referenced_works") or []),
|
|
50
|
+
related_works=list(work.get("related_works") or []),
|
|
51
|
+
topics=_topics_from_openalex(work),
|
|
52
|
+
oa_status=oa.get("oa_status"),
|
|
53
|
+
oa_url=best_oa.get("pdf_url") or best_oa.get("landing_page_url"),
|
|
54
|
+
provenance=[provenance] if provenance else [],
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def from_crossref_item(item: dict, *, provenance: str = "crossref") -> Record:
|
|
59
|
+
titles = item.get("title") or []
|
|
60
|
+
containers = item.get("container-title") or []
|
|
61
|
+
authors = []
|
|
62
|
+
for a in item.get("author") or []:
|
|
63
|
+
nm = " ".join(x for x in (a.get("given"), a.get("family")) if x) or a.get("name")
|
|
64
|
+
if nm:
|
|
65
|
+
authors.append(nm)
|
|
66
|
+
issued = ((item.get("issued") or {}).get("date-parts") or [[None]])[0]
|
|
67
|
+
return Record(
|
|
68
|
+
doi=item.get("DOI"),
|
|
69
|
+
title=titles[0] if titles else None,
|
|
70
|
+
authors=authors,
|
|
71
|
+
year=issued[0] if issued else None,
|
|
72
|
+
venue=containers[0] if containers else None,
|
|
73
|
+
type=item.get("type"),
|
|
74
|
+
cited_by_count=item.get("is-referenced-by-count"),
|
|
75
|
+
provenance=[provenance],
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def from_s2_paper(paper: dict, *, provenance: str = "semantic_scholar") -> Record:
|
|
80
|
+
ext = paper.get("externalIds") or {}
|
|
81
|
+
tldr = (paper.get("tldr") or {}).get("text")
|
|
82
|
+
abstract = paper.get("abstract") or tldr
|
|
83
|
+
return Record(
|
|
84
|
+
doi=ext.get("DOI"),
|
|
85
|
+
title=paper.get("title"),
|
|
86
|
+
year=paper.get("year"),
|
|
87
|
+
venue=paper.get("venue"),
|
|
88
|
+
abstract=abstract,
|
|
89
|
+
abstract_source=("semantic_scholar" if paper.get("abstract")
|
|
90
|
+
else ("s2_tldr" if tldr else None)),
|
|
91
|
+
cited_by_count=paper.get("citationCount"),
|
|
92
|
+
topics=list(paper.get("fieldsOfStudy") or []),
|
|
93
|
+
provenance=[provenance],
|
|
94
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""CORE client. Key is optional but raises the rate limit from 10/min to 150/min."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
from scholarlib.http.base import BaseClient
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
BASE_URL = "https://api.core.ac.uk/v3"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class COREClient(BaseClient):
|
|
16
|
+
name = "core"
|
|
17
|
+
base_url = BASE_URL
|
|
18
|
+
|
|
19
|
+
def __init__(self, api_key: Optional[str] = None, **kw):
|
|
20
|
+
super().__init__(**kw)
|
|
21
|
+
self.api_key = api_key
|
|
22
|
+
if not api_key:
|
|
23
|
+
logger.debug("CORE: no API key — rate limit is 10/min instead of 150/min")
|
|
24
|
+
|
|
25
|
+
def _auth_headers(self) -> dict:
|
|
26
|
+
return {"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}
|
|
27
|
+
|
|
28
|
+
def search_works(self, query: str, limit: int = 10) -> list[dict]:
|
|
29
|
+
data = self.get("/search/works", {"q": query, "limit": limit})
|
|
30
|
+
return (data or {}).get("results", [])
|
|
31
|
+
|
|
32
|
+
def get_work_by_doi(self, doi: str) -> Optional[dict]:
|
|
33
|
+
d = str(doi).replace("https://doi.org/", "").strip()
|
|
34
|
+
results = self.search_works(f'doi:"{d}"', limit=1)
|
|
35
|
+
return results[0] if results else None
|
|
36
|
+
|
|
37
|
+
def get_abstract(self, doi: str) -> Optional[str]:
|
|
38
|
+
w = self.get_work_by_doi(doi)
|
|
39
|
+
if not w:
|
|
40
|
+
return None
|
|
41
|
+
return w.get("abstract") or w.get("description") or None
|