medsci-skills 5.0.0 → 5.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/metadata/distribution_files.json +47 -12
- package/metadata/distribution_manifest.json +1 -1
- package/package.json +1 -1
- package/skills/fulltext-retrieval/SKILL.md +52 -5
- package/skills/fulltext-retrieval/fetch_oa.py +293 -74
- package/skills/fulltext-retrieval/fetch_oa_report_challenge/expected/projection.json +14 -0
- package/skills/fulltext-retrieval/fetch_oa_report_challenge/extracted_text.json +4 -0
- package/skills/fulltext-retrieval/fetch_oa_report_challenge/results.json +6 -0
- package/skills/fulltext-retrieval/fetch_oa_report_challenge/run_challenge.py +86 -0
- package/skills/fulltext-retrieval/fetch_oa_report_challenge/verify.sh +6 -0
- package/skills/fulltext-retrieval/fetch_oa_report_challenge/worklist.tsv +5 -0
- package/skills/fulltext-retrieval/references/find_available_pdf.js +82 -0
- package/skills/fulltext-retrieval/skill.yml +7 -2
- package/skills/lit-sync/SKILL.md +86 -6
- package/skills/lit-sync/skill.yml +3 -1
- package/skills/search-lit/SKILL.md +22 -67
|
@@ -1488,23 +1488,58 @@
|
|
|
1488
1488
|
},
|
|
1489
1489
|
{
|
|
1490
1490
|
"path": "skills/fulltext-retrieval/SKILL.md",
|
|
1491
|
-
"size":
|
|
1492
|
-
"sha256": "
|
|
1491
|
+
"size": 8248,
|
|
1492
|
+
"sha256": "d97c19a64d92f66e3c8a1445d893c0f55364f21dc99404fb6f012a3566556e73"
|
|
1493
1493
|
},
|
|
1494
1494
|
{
|
|
1495
1495
|
"path": "skills/fulltext-retrieval/fetch_oa.py",
|
|
1496
|
-
"size":
|
|
1497
|
-
"sha256": "
|
|
1496
|
+
"size": 24503,
|
|
1497
|
+
"sha256": "768f95a30737e67b7f9edee72fe8815a795c9f83451b46551bbe25fc36094f4b"
|
|
1498
|
+
},
|
|
1499
|
+
{
|
|
1500
|
+
"path": "skills/fulltext-retrieval/fetch_oa_report_challenge/expected/projection.json",
|
|
1501
|
+
"size": 522,
|
|
1502
|
+
"sha256": "f24bdb507f66932c0429acefc171abf5221bd644cacdeba5cd7b5c07670b04aa"
|
|
1503
|
+
},
|
|
1504
|
+
{
|
|
1505
|
+
"path": "skills/fulltext-retrieval/fetch_oa_report_challenge/extracted_text.json",
|
|
1506
|
+
"size": 351,
|
|
1507
|
+
"sha256": "9b512bbea3780727e79355dd4b3689be182f086d4afb4184737524449fc44e4d"
|
|
1508
|
+
},
|
|
1509
|
+
{
|
|
1510
|
+
"path": "skills/fulltext-retrieval/fetch_oa_report_challenge/results.json",
|
|
1511
|
+
"size": 175,
|
|
1512
|
+
"sha256": "e46d52e6a9c038ebd5cb8ce89a5961f32c8365e80d189d3ee80bed021a60296a"
|
|
1513
|
+
},
|
|
1514
|
+
{
|
|
1515
|
+
"path": "skills/fulltext-retrieval/fetch_oa_report_challenge/run_challenge.py",
|
|
1516
|
+
"size": 3138,
|
|
1517
|
+
"sha256": "a3c194bb31f319bdd011d4436e5de22cfcc79b7979fbcba4883ed370e4acbeb4"
|
|
1518
|
+
},
|
|
1519
|
+
{
|
|
1520
|
+
"path": "skills/fulltext-retrieval/fetch_oa_report_challenge/verify.sh",
|
|
1521
|
+
"size": 286,
|
|
1522
|
+
"sha256": "f96687fa9d1b213734810dcdfa8126a2efd1af106f5d1cdeb43d58b431ab915b"
|
|
1523
|
+
},
|
|
1524
|
+
{
|
|
1525
|
+
"path": "skills/fulltext-retrieval/fetch_oa_report_challenge/worklist.tsv",
|
|
1526
|
+
"size": 326,
|
|
1527
|
+
"sha256": "d1c4c906052be198275155ee3e9ecf29b04ec8bdfeef97816ca15aa8cefa61fa"
|
|
1498
1528
|
},
|
|
1499
1529
|
{
|
|
1500
1530
|
"path": "skills/fulltext-retrieval/pdf_to_md.py",
|
|
1501
1531
|
"size": 5398,
|
|
1502
1532
|
"sha256": "16ba8c61db254b4b356d85686946351daa5ea650d70dca2fd0d5677851c6df4d"
|
|
1503
1533
|
},
|
|
1534
|
+
{
|
|
1535
|
+
"path": "skills/fulltext-retrieval/references/find_available_pdf.js",
|
|
1536
|
+
"size": 3057,
|
|
1537
|
+
"sha256": "04dc6d13d0bb7c43679e5386f1788edee6b187b2503cd9e250ae4ea63e11cec0"
|
|
1538
|
+
},
|
|
1504
1539
|
{
|
|
1505
1540
|
"path": "skills/fulltext-retrieval/skill.yml",
|
|
1506
|
-
"size":
|
|
1507
|
-
"sha256": "
|
|
1541
|
+
"size": 2420,
|
|
1542
|
+
"sha256": "7a457e5f5fde09f57a8ef9d2207dbaab241e56a9bca59ba577d26351bf225346"
|
|
1508
1543
|
},
|
|
1509
1544
|
{
|
|
1510
1545
|
"path": "skills/generate-codebook/SKILL.md",
|
|
@@ -1563,8 +1598,8 @@
|
|
|
1563
1598
|
},
|
|
1564
1599
|
{
|
|
1565
1600
|
"path": "skills/lit-sync/SKILL.md",
|
|
1566
|
-
"size":
|
|
1567
|
-
"sha256": "
|
|
1601
|
+
"size": 21126,
|
|
1602
|
+
"sha256": "4ba2d2b6dd4704803c0922993401b837d4135ac1d655d6064229315537a8e818"
|
|
1568
1603
|
},
|
|
1569
1604
|
{
|
|
1570
1605
|
"path": "skills/lit-sync/references/locale/ko/note_templates.md",
|
|
@@ -1573,8 +1608,8 @@
|
|
|
1573
1608
|
},
|
|
1574
1609
|
{
|
|
1575
1610
|
"path": "skills/lit-sync/skill.yml",
|
|
1576
|
-
"size":
|
|
1577
|
-
"sha256": "
|
|
1611
|
+
"size": 2780,
|
|
1612
|
+
"sha256": "a04d160e9fc6382c783940b737c9d9348db45e34ea80c69ef905be0a3c357e34"
|
|
1578
1613
|
},
|
|
1579
1614
|
{
|
|
1580
1615
|
"path": "skills/ma-scout/SKILL.md",
|
|
@@ -3238,8 +3273,8 @@
|
|
|
3238
3273
|
},
|
|
3239
3274
|
{
|
|
3240
3275
|
"path": "skills/search-lit/SKILL.md",
|
|
3241
|
-
"size":
|
|
3242
|
-
"sha256": "
|
|
3276
|
+
"size": 20281,
|
|
3277
|
+
"sha256": "608644e83ffc357f3b43b65cf7c439c9381f5948c51798672929f7a38aea5024"
|
|
3243
3278
|
},
|
|
3244
3279
|
{
|
|
3245
3280
|
"path": "skills/search-lit/references/parse_pubmed.py",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "medsci-skills",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.1.0",
|
|
4
4
|
"description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"homepage": "https://github.com/Aperivue/medsci-skills#readme",
|
|
@@ -13,10 +13,10 @@ Batch download open-access full-text PDFs from a DOI list using legitimate OA AP
|
|
|
13
13
|
## Pipeline
|
|
14
14
|
|
|
15
15
|
```
|
|
16
|
-
DOI
|
|
16
|
+
DOI → arXiv (10.48550/arXiv.* DOIs) → Unpaywall → PMC (Europe PMC / OA FTP / web) → OpenAlex → Crossref → landing page
|
|
17
17
|
```
|
|
18
18
|
|
|
19
|
-
Each DOI goes through these sources in order until a valid PDF (≥10 KB, `%PDF-` header) is found.
|
|
19
|
+
Each DOI goes through these sources in order until a valid PDF (≥10 KB, `%PDF-` header) is found. arXiv DOIs (`10.48550/arXiv.2401.01234`, version suffixes, old-style `hep-th/9901001`, or a bare `arXiv:` id) resolve directly to the arXiv PDF first.
|
|
20
20
|
|
|
21
21
|
## Quick Start
|
|
22
22
|
|
|
@@ -43,13 +43,20 @@ python fetch_oa.py dois.txt -o pdfs/ -e your@email.com --verbose
|
|
|
43
43
|
10.1002/mp.12524
|
|
44
44
|
```
|
|
45
45
|
|
|
46
|
-
**TSV with header** — must contain a `DOI` column
|
|
46
|
+
**TSV / CSV with header** — must contain a `DOI` column; optional `PMID` and `Title` columns:
|
|
47
47
|
```tsv
|
|
48
48
|
ID Title DOI PMID Year
|
|
49
49
|
1 Some paper 10.1007/s00330-010-1783-x 20628747 2010
|
|
50
50
|
```
|
|
51
51
|
|
|
52
|
-
|
|
52
|
+
**Markdown table** — a pipe table with a `DOI` column also works:
|
|
53
|
+
```markdown
|
|
54
|
+
| DOI | PMID | Title |
|
|
55
|
+
|-----|------|-------|
|
|
56
|
+
| 10.1007/s00330-010-1783-x | 20628747 | Some paper |
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
When a PMID is available, the PMC lookup is more reliable (PMID → PMCID conversion). When a `Title` column is present, downloaded PDFs get a best-effort title cross-check (see *Retrieval report* below).
|
|
53
60
|
|
|
54
61
|
## PMC Download (JS-Challenge Resistant)
|
|
55
62
|
|
|
@@ -82,8 +89,48 @@ curl -s "https://www.ncbi.nlm.nih.gov/pmc/utils/idconv/v1.0/?ids=${DOI}&format=j
|
|
|
82
89
|
## Output
|
|
83
90
|
|
|
84
91
|
- PDFs saved as `{DOI_safe}.pdf` (slashes replaced with underscores)
|
|
92
|
+
- `pdfs/retrieval_report.json` — structured per-DOI report (see below)
|
|
85
93
|
- `manual_needed.txt` — DOIs that could not be retrieved via OA
|
|
86
|
-
- Summary with OA/PMC/fail/skip counts
|
|
94
|
+
- Summary with arXiv/OA/PMC/fail/skip counts
|
|
95
|
+
|
|
96
|
+
## Retrieval report (`--report`)
|
|
97
|
+
|
|
98
|
+
Every run writes a structured report (default `<output>/retrieval_report.json`,
|
|
99
|
+
override with `--report PATH`):
|
|
100
|
+
|
|
101
|
+
```json
|
|
102
|
+
{
|
|
103
|
+
"schema_version": 1,
|
|
104
|
+
"generated_by": "fetch_oa.py",
|
|
105
|
+
"counts": {"total": 10, "retrieved": 6, "not_retrieved": 4, "title_mismatch": 1},
|
|
106
|
+
"items": [
|
|
107
|
+
{"doi": "10.1007/...", "pmid": "20628747", "title": "...",
|
|
108
|
+
"status": "oa", "source": "unpaywall", "file": "10.1007_....pdf",
|
|
109
|
+
"size_bytes": 482113, "title_match": "match"}
|
|
110
|
+
]
|
|
111
|
+
}
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
- `status` ∈ `arxiv | oa | pmc | skip | fail`; `source` names the resolver that succeeded.
|
|
115
|
+
- `title_match` ∈ `match | mismatch | unavailable` (tri-state). It is **best-effort**:
|
|
116
|
+
it needs a `Title` column **and** `pdftotext` (poppler). When either is missing it is
|
|
117
|
+
`unavailable`; a `mismatch` is **flagged** for review and **never** auto-rejects a PDF
|
|
118
|
+
(guards against a publisher serving a wrong/redirect PDF that still passes the `%PDF-` check).
|
|
119
|
+
|
|
120
|
+
## Attach PDFs into Zotero ("Find Available PDF")
|
|
121
|
+
|
|
122
|
+
OA-only resolvers miss paywalled-but-licensed papers. To attach full text **inside
|
|
123
|
+
Zotero** at a much higher yield, use `references/find_available_pdf.js` — a user-run
|
|
124
|
+
snippet for Zotero's *Tools → Developer → Run JavaScript*. It triggers Zotero's own
|
|
125
|
+
`addAvailablePDF` / `addAvailablePDFs` and therefore reuses **your** OpenURL resolver /
|
|
126
|
+
institutional proxy config; **no credentials, proxy hosts, or institutional identifiers
|
|
127
|
+
are hard-coded or leave your Zotero client**. The no-code equivalent is right-click →
|
|
128
|
+
"Find Available PDF".
|
|
129
|
+
|
|
130
|
+
This path is **user-initiated** and depends on your live Zotero session, so its results
|
|
131
|
+
are recorded manually (not reproducible CI evidence). `/lit-sync` Phase 2.7 orchestrates
|
|
132
|
+
both routes (disk OA via this script + in-library via the snippet) and reconciles them in
|
|
133
|
+
a report.
|
|
87
134
|
|
|
88
135
|
## Requirements
|
|
89
136
|
|
|
@@ -2,20 +2,28 @@
|
|
|
2
2
|
"""
|
|
3
3
|
Open-access full-text PDF batch retrieval.
|
|
4
4
|
|
|
5
|
-
Pipeline:
|
|
6
|
-
OpenAlex → Crossref →
|
|
5
|
+
Pipeline: arXiv (for 10.48550/arXiv.* DOIs) → Unpaywall →
|
|
6
|
+
PMC (Europe PMC REST / OA FTP / web) → OpenAlex → Crossref →
|
|
7
|
+
landing-page scrape.
|
|
7
8
|
|
|
8
9
|
Usage:
|
|
9
10
|
python fetch_oa.py dois.txt --output pdfs/ --email user@example.com
|
|
10
|
-
python fetch_oa.py
|
|
11
|
+
python fetch_oa.py worklist.tsv -o pdfs/ -e user@example.com --verbose
|
|
12
|
+
python fetch_oa.py worklist.csv -o pdfs/ -e user@example.com --report pdfs/retrieval_report.json
|
|
13
|
+
|
|
14
|
+
Worklist formats: plain DOI-per-line, or TSV/CSV/Markdown-table with a DOI
|
|
15
|
+
column (optional PMID and Title columns). A Title column enables a best-effort
|
|
16
|
+
title cross-check (via `pdftotext` if installed) that flags mislabeled PDFs.
|
|
11
17
|
"""
|
|
12
18
|
|
|
13
19
|
import argparse
|
|
20
|
+
import csv
|
|
21
|
+
import io
|
|
14
22
|
import json
|
|
15
23
|
import logging
|
|
16
|
-
import os
|
|
17
24
|
import re
|
|
18
|
-
import
|
|
25
|
+
import shutil
|
|
26
|
+
import subprocess
|
|
19
27
|
import time
|
|
20
28
|
import urllib.error
|
|
21
29
|
import urllib.parse
|
|
@@ -25,6 +33,9 @@ from pathlib import Path
|
|
|
25
33
|
|
|
26
34
|
MIN_PDF_BYTES = 10 * 1024
|
|
27
35
|
USER_AGENT = "medsci-skills/1.0"
|
|
36
|
+
REPORT_SCHEMA_VERSION = 1
|
|
37
|
+
TITLE_MATCH_THRESHOLD = 0.6
|
|
38
|
+
RETRIEVED_STATUSES = ("oa", "pmc", "arxiv", "skip")
|
|
28
39
|
|
|
29
40
|
log = logging.getLogger("fetch_oa")
|
|
30
41
|
|
|
@@ -38,6 +49,11 @@ def _ua(email: str) -> str:
|
|
|
38
49
|
return f"{USER_AGENT} (mailto:{email})"
|
|
39
50
|
|
|
40
51
|
|
|
52
|
+
def safe_doi_name(doi: str) -> str:
|
|
53
|
+
"""Filesystem-safe filename stem for a DOI."""
|
|
54
|
+
return re.sub(r"[^\w\-.]", "_", doi)
|
|
55
|
+
|
|
56
|
+
|
|
41
57
|
def is_valid_pdf(data: bytes) -> bool:
|
|
42
58
|
return data.startswith(b"%PDF-") and len(data) >= MIN_PDF_BYTES
|
|
43
59
|
|
|
@@ -68,6 +84,84 @@ def existing_pdf_ok(path: Path) -> bool:
|
|
|
68
84
|
return False
|
|
69
85
|
|
|
70
86
|
|
|
87
|
+
# ============================================================
|
|
88
|
+
# Title cross-check (pure, offline-testable)
|
|
89
|
+
# ============================================================
|
|
90
|
+
|
|
91
|
+
def normalize_title(text: str) -> str:
|
|
92
|
+
"""Lowercase, strip punctuation, collapse whitespace."""
|
|
93
|
+
text = (text or "").lower()
|
|
94
|
+
text = re.sub(r"[^a-z0-9\s]", " ", text)
|
|
95
|
+
return " ".join(text.split())
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def title_overlap(expected_title: str, extracted_text: str) -> float:
|
|
99
|
+
"""Fraction of meaningful expected-title tokens present in extracted text.
|
|
100
|
+
|
|
101
|
+
Tokens of length <= 2 are dropped as near-stopwords. Returns 0.0 when the
|
|
102
|
+
expected title has no usable tokens.
|
|
103
|
+
"""
|
|
104
|
+
expected = {t for t in normalize_title(expected_title).split() if len(t) > 2}
|
|
105
|
+
if not expected:
|
|
106
|
+
return 0.0
|
|
107
|
+
got = set(normalize_title(extracted_text).split())
|
|
108
|
+
return len(expected & got) / len(expected)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def classify_title_match(expected_title: str, extracted_text,
|
|
112
|
+
threshold: float = TITLE_MATCH_THRESHOLD) -> str:
|
|
113
|
+
"""Tri-state title check: 'match' | 'mismatch' | 'unavailable'.
|
|
114
|
+
|
|
115
|
+
'unavailable' when no expected title is supplied or no extracted text is
|
|
116
|
+
available (e.g. pdftotext absent). A low overlap is 'mismatch' — flagged,
|
|
117
|
+
never used to auto-reject a downloaded PDF.
|
|
118
|
+
"""
|
|
119
|
+
if not expected_title or not extracted_text:
|
|
120
|
+
return "unavailable"
|
|
121
|
+
return "match" if title_overlap(expected_title, extracted_text) >= threshold else "mismatch"
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def extract_pdf_text(path: Path, max_pages: int = 2) -> str | None:
|
|
125
|
+
"""Best-effort first-page text via `pdftotext` (poppler). None if unavailable."""
|
|
126
|
+
if not shutil.which("pdftotext"):
|
|
127
|
+
return None
|
|
128
|
+
try:
|
|
129
|
+
out = subprocess.run(
|
|
130
|
+
["pdftotext", "-f", "1", "-l", str(max_pages), str(path), "-"],
|
|
131
|
+
capture_output=True, timeout=20,
|
|
132
|
+
)
|
|
133
|
+
if out.returncode == 0:
|
|
134
|
+
return out.stdout.decode("utf-8", errors="ignore")
|
|
135
|
+
except (OSError, subprocess.SubprocessError):
|
|
136
|
+
pass
|
|
137
|
+
return None
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
# ============================================================
|
|
141
|
+
# arXiv (direct, for 10.48550/arXiv.* DOIs)
|
|
142
|
+
# ============================================================
|
|
143
|
+
|
|
144
|
+
_ARXIV_DOI_RE = re.compile(r"^10\.48550/arxiv\.(.+)$", re.IGNORECASE)
|
|
145
|
+
_ARXIV_ID_RE = re.compile(r"^arxiv:(.+)$", re.IGNORECASE)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def arxiv_id_from_doi(doi: str) -> str | None:
|
|
149
|
+
"""Extract an arXiv ID from a DataCite arXiv DOI or a bare arXiv: id.
|
|
150
|
+
|
|
151
|
+
Handles new-style (2401.01234, 2401.01234v2) and old-style
|
|
152
|
+
(hep-th/9901001) identifiers; version suffix preserved when present.
|
|
153
|
+
"""
|
|
154
|
+
s = (doi or "").strip()
|
|
155
|
+
m = _ARXIV_DOI_RE.match(s) or _ARXIV_ID_RE.match(s)
|
|
156
|
+
return m.group(1).strip() if m else None
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def arxiv_pdf_url(doi: str) -> str | None:
|
|
160
|
+
"""Direct arXiv PDF URL for an arXiv DOI/ID (None if not an arXiv id)."""
|
|
161
|
+
aid = arxiv_id_from_doi(doi)
|
|
162
|
+
return f"https://arxiv.org/pdf/{aid}" if aid else None
|
|
163
|
+
|
|
164
|
+
|
|
71
165
|
# ============================================================
|
|
72
166
|
# 1. Unpaywall
|
|
73
167
|
# ============================================================
|
|
@@ -270,41 +364,34 @@ def download_pdf(url: str, outpath: Path, email: str) -> bool:
|
|
|
270
364
|
# 5. Main pipeline
|
|
271
365
|
# ============================================================
|
|
272
366
|
|
|
273
|
-
def gather_candidates(doi: str, email: str) -> list[str]:
|
|
274
|
-
"""Collect OA PDF candidate URLs from multiple sources."""
|
|
275
|
-
urls: list[str] = []
|
|
276
|
-
|
|
277
|
-
def add(v: str | None):
|
|
278
|
-
if v and v not in urls:
|
|
279
|
-
urls.append(v)
|
|
280
|
-
|
|
281
|
-
add(unpaywall_lookup(doi, email))
|
|
282
|
-
for v in openalex_lookup(doi, email):
|
|
283
|
-
add(v)
|
|
284
|
-
for v in crossref_lookup(doi, email):
|
|
285
|
-
add(v)
|
|
286
|
-
add(f"https://doi.org/{doi}")
|
|
287
|
-
return urls
|
|
288
|
-
|
|
289
|
-
|
|
290
367
|
def process_doi(doi: str, outdir: Path, email: str,
|
|
291
|
-
pmid: str = "") -> str:
|
|
292
|
-
"""Try to download a PDF for one DOI.
|
|
293
|
-
|
|
294
|
-
|
|
368
|
+
pmid: str = "") -> tuple[str, str]:
|
|
369
|
+
"""Try to download a PDF for one DOI.
|
|
370
|
+
|
|
371
|
+
Returns (status, source):
|
|
372
|
+
status ∈ {"arxiv", "oa", "pmc", "skip", "fail"}
|
|
373
|
+
source identifies the resolver that succeeded (e.g. "unpaywall", "pmc",
|
|
374
|
+
"openalex", "crossref", "landing", "arxiv", "existing", "").
|
|
375
|
+
"""
|
|
376
|
+
outpath = outdir / f"{safe_doi_name(doi)}.pdf"
|
|
295
377
|
|
|
296
378
|
if existing_pdf_ok(outpath):
|
|
297
|
-
return "skip"
|
|
379
|
+
return ("skip", "existing")
|
|
298
380
|
|
|
299
381
|
# Remove stale stub
|
|
300
382
|
if outpath.exists():
|
|
301
383
|
outpath.unlink(missing_ok=True)
|
|
302
384
|
|
|
385
|
+
# Step 0: arXiv direct (for 10.48550/arXiv.* DOIs)
|
|
386
|
+
ax_url = arxiv_pdf_url(doi)
|
|
387
|
+
if ax_url and download_pdf(ax_url, outpath, email):
|
|
388
|
+
return ("arxiv", "arxiv")
|
|
389
|
+
|
|
303
390
|
# Step 1: Unpaywall direct PDF URL (fastest path)
|
|
304
391
|
uw_url = unpaywall_lookup(doi, email)
|
|
305
392
|
if uw_url and ".pdf" in uw_url.lower():
|
|
306
393
|
if download_pdf(uw_url, outpath, email):
|
|
307
|
-
return "oa"
|
|
394
|
+
return ("oa", "unpaywall")
|
|
308
395
|
time.sleep(0.3)
|
|
309
396
|
|
|
310
397
|
# Step 2: PMC (try before slow landing-page scraping)
|
|
@@ -312,59 +399,158 @@ def process_doi(doi: str, outdir: Path, email: str,
|
|
|
312
399
|
if not pmcid:
|
|
313
400
|
pmcid = id_to_pmcid(doi, email)
|
|
314
401
|
if pmcid and download_pmc_pdf(pmcid, outpath, email):
|
|
315
|
-
return "pmc"
|
|
402
|
+
return ("pmc", "pmc")
|
|
316
403
|
|
|
317
404
|
# Step 3: OA candidates from OpenAlex, Crossref, landing pages
|
|
318
|
-
candidates: list[str] = []
|
|
319
|
-
|
|
320
|
-
|
|
405
|
+
candidates: list[tuple[str, str]] = []
|
|
406
|
+
seen: set[str] = set()
|
|
407
|
+
|
|
408
|
+
def add(source: str, url: str | None):
|
|
409
|
+
if url and url not in seen:
|
|
410
|
+
seen.add(url)
|
|
411
|
+
candidates.append((source, url))
|
|
412
|
+
|
|
413
|
+
add("unpaywall", uw_url)
|
|
321
414
|
for v in openalex_lookup(doi, email):
|
|
322
|
-
|
|
323
|
-
candidates.append(v)
|
|
415
|
+
add("openalex", v)
|
|
324
416
|
for v in crossref_lookup(doi, email):
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
candidates.append(f"https://doi.org/{doi}")
|
|
417
|
+
add("crossref", v)
|
|
418
|
+
add("landing", f"https://doi.org/{doi}")
|
|
328
419
|
|
|
329
|
-
for url in candidates:
|
|
420
|
+
for source, url in candidates:
|
|
330
421
|
if ".pdf" in url.lower():
|
|
331
422
|
ok = download_pdf(url, outpath, email)
|
|
332
423
|
else:
|
|
333
424
|
ok = download_from_landing(url, outpath, email)
|
|
334
425
|
if ok:
|
|
335
|
-
return "oa"
|
|
426
|
+
return ("oa", source)
|
|
336
427
|
time.sleep(0.3)
|
|
337
428
|
|
|
338
|
-
return "fail"
|
|
429
|
+
return ("fail", "")
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def build_report(records: list[dict], results: dict[str, tuple[str, str]],
|
|
433
|
+
outdir: Path, extracted_text_by_doi: dict[str, str] | None = None,
|
|
434
|
+
threshold: float = TITLE_MATCH_THRESHOLD) -> dict:
|
|
435
|
+
"""Assemble a deterministic retrieval report (no network, no I/O writes).
|
|
436
|
+
|
|
437
|
+
records: list of {"doi", "pmid", "title"}.
|
|
438
|
+
results: doi -> (status, source) as returned by process_doi.
|
|
439
|
+
outdir: directory where PDFs were written (used for file/size lookup).
|
|
440
|
+
extracted_text_by_doi: optional doi -> first-page text for title cross-check.
|
|
441
|
+
"""
|
|
442
|
+
extracted_text_by_doi = extracted_text_by_doi or {}
|
|
443
|
+
items = []
|
|
444
|
+
for rec in records:
|
|
445
|
+
doi = rec["doi"]
|
|
446
|
+
status, source = results.get(doi, ("fail", ""))
|
|
447
|
+
path = outdir / f"{safe_doi_name(doi)}.pdf"
|
|
448
|
+
have_file = status in RETRIEVED_STATUSES and path.exists()
|
|
449
|
+
size = path.stat().st_size if have_file else 0
|
|
450
|
+
if have_file:
|
|
451
|
+
title_match = classify_title_match(
|
|
452
|
+
rec.get("title", ""), extracted_text_by_doi.get(doi), threshold)
|
|
453
|
+
else:
|
|
454
|
+
title_match = "unavailable"
|
|
455
|
+
items.append({
|
|
456
|
+
"doi": doi,
|
|
457
|
+
"pmid": rec.get("pmid", ""),
|
|
458
|
+
"title": rec.get("title", ""),
|
|
459
|
+
"status": status,
|
|
460
|
+
"source": source,
|
|
461
|
+
"file": path.name if have_file else "",
|
|
462
|
+
"size_bytes": size,
|
|
463
|
+
"title_match": title_match,
|
|
464
|
+
})
|
|
465
|
+
|
|
466
|
+
retrieved = [i for i in items if i["status"] in RETRIEVED_STATUSES]
|
|
467
|
+
not_retrieved = [i for i in items if i["status"] == "fail"]
|
|
468
|
+
return {
|
|
469
|
+
"schema_version": REPORT_SCHEMA_VERSION,
|
|
470
|
+
"generated_by": "fetch_oa.py",
|
|
471
|
+
"counts": {
|
|
472
|
+
"total": len(items),
|
|
473
|
+
"retrieved": len(retrieved),
|
|
474
|
+
"not_retrieved": len(not_retrieved),
|
|
475
|
+
"title_mismatch": sum(1 for i in items if i["title_match"] == "mismatch"),
|
|
476
|
+
},
|
|
477
|
+
"items": items,
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def _norm_key(key: str) -> str:
|
|
482
|
+
return (key or "").strip().lstrip("#").strip().lower()
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def _records_from_dictrows(rows) -> list[dict]:
|
|
486
|
+
records = []
|
|
487
|
+
for row in rows:
|
|
488
|
+
rec = {"doi": "", "pmid": "", "title": ""}
|
|
489
|
+
for k, v in row.items():
|
|
490
|
+
nk = _norm_key(k)
|
|
491
|
+
if nk in rec:
|
|
492
|
+
rec[nk] = (v or "").strip()
|
|
493
|
+
if rec["doi"]:
|
|
494
|
+
records.append(rec)
|
|
495
|
+
return records
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def _records_from_markdown(lines: list[str]) -> list[dict]:
|
|
499
|
+
pipe_rows = [ln for ln in lines if ln.strip().startswith("|")]
|
|
500
|
+
if not pipe_rows:
|
|
501
|
+
return []
|
|
502
|
+
|
|
503
|
+
def cells(line: str) -> list[str]:
|
|
504
|
+
return [c.strip() for c in line.strip().strip("|").split("|")]
|
|
505
|
+
|
|
506
|
+
header = [_norm_key(c) for c in cells(pipe_rows[0])]
|
|
507
|
+
records = []
|
|
508
|
+
for line in pipe_rows[1:]:
|
|
509
|
+
c = cells(line)
|
|
510
|
+
# Skip the |---|---| separator row
|
|
511
|
+
if c and all(set(x) <= set("-: ") for x in c):
|
|
512
|
+
continue
|
|
513
|
+
row = dict(zip(header, c))
|
|
514
|
+
doi = (row.get("doi") or "").strip()
|
|
515
|
+
if doi:
|
|
516
|
+
records.append({
|
|
517
|
+
"doi": doi,
|
|
518
|
+
"pmid": (row.get("pmid") or "").strip(),
|
|
519
|
+
"title": (row.get("title") or "").strip(),
|
|
520
|
+
})
|
|
521
|
+
return records
|
|
339
522
|
|
|
340
523
|
|
|
341
524
|
def read_doi_file(path: Path) -> list[dict]:
|
|
342
|
-
"""Read
|
|
525
|
+
"""Read a worklist of DOIs.
|
|
526
|
+
|
|
527
|
+
Supports: plain DOI-per-line; TSV/CSV with a DOI header (optional PMID,
|
|
528
|
+
Title columns); and a Markdown pipe table with a DOI column. Each record is
|
|
529
|
+
{"doi", "pmid", "title"}.
|
|
530
|
+
"""
|
|
531
|
+
text = Path(path).read_text(encoding="utf-8")
|
|
532
|
+
lines = text.splitlines()
|
|
533
|
+
first = next((ln for ln in lines
|
|
534
|
+
if ln.strip() and not ln.strip().startswith("#")), "")
|
|
535
|
+
low = first.lower()
|
|
536
|
+
|
|
537
|
+
# Markdown pipe table with a DOI column
|
|
538
|
+
if first.strip().startswith("|") and "doi" in low:
|
|
539
|
+
return _records_from_markdown(lines)
|
|
540
|
+
|
|
541
|
+
# Delimited (TSV or CSV) with a DOI header
|
|
542
|
+
if "doi" in low and ("\t" in first or "," in first):
|
|
543
|
+
delimiter = "\t" if "\t" in first else ","
|
|
544
|
+
body = "\n".join(ln for ln in lines if not ln.strip().startswith("#"))
|
|
545
|
+
reader = csv.DictReader(io.StringIO(body), delimiter=delimiter)
|
|
546
|
+
return _records_from_dictrows(reader)
|
|
547
|
+
|
|
548
|
+
# Plain text: one DOI per line
|
|
343
549
|
records = []
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
# TSV with header containing DOI column
|
|
349
|
-
if "\t" in first_line and "doi" in first_line.lower():
|
|
350
|
-
import csv
|
|
351
|
-
reader = csv.DictReader(f, delimiter="\t")
|
|
352
|
-
for row in reader:
|
|
353
|
-
doi = ""
|
|
354
|
-
pmid = ""
|
|
355
|
-
for k, v in row.items():
|
|
356
|
-
if k.lower().strip() == "doi":
|
|
357
|
-
doi = (v or "").strip()
|
|
358
|
-
elif k.lower().strip() == "pmid":
|
|
359
|
-
pmid = (v or "").strip()
|
|
360
|
-
if doi:
|
|
361
|
-
records.append({"doi": doi, "pmid": pmid})
|
|
362
|
-
else:
|
|
363
|
-
# Plain text: one DOI per line
|
|
364
|
-
for line in f:
|
|
365
|
-
line = line.strip()
|
|
366
|
-
if line and not line.startswith("#"):
|
|
367
|
-
records.append({"doi": line, "pmid": ""})
|
|
550
|
+
for line in lines:
|
|
551
|
+
line = line.strip()
|
|
552
|
+
if line and not line.startswith("#"):
|
|
553
|
+
records.append({"doi": line, "pmid": "", "title": ""})
|
|
368
554
|
return records
|
|
369
555
|
|
|
370
556
|
|
|
@@ -372,11 +558,15 @@ def main():
|
|
|
372
558
|
parser = argparse.ArgumentParser(
|
|
373
559
|
description="Batch download open-access PDFs by DOI.")
|
|
374
560
|
parser.add_argument("input", type=Path,
|
|
375
|
-
help="
|
|
561
|
+
help="Worklist: DOIs (one per line) or TSV/CSV/Markdown "
|
|
562
|
+
"with a DOI column (optional PMID, Title)")
|
|
376
563
|
parser.add_argument("-o", "--output", type=Path, default=Path("pdfs"),
|
|
377
564
|
help="Output directory (default: pdfs/)")
|
|
378
565
|
parser.add_argument("-e", "--email", required=True,
|
|
379
566
|
help="Contact email (required by Unpaywall TOS)")
|
|
567
|
+
parser.add_argument("--report", type=Path, default=None,
|
|
568
|
+
help="Path for the JSON retrieval report "
|
|
569
|
+
"(default: <output>/retrieval_report.json)")
|
|
380
570
|
parser.add_argument("-v", "--verbose", action="store_true",
|
|
381
571
|
help="Show debug messages")
|
|
382
572
|
args = parser.parse_args()
|
|
@@ -387,33 +577,62 @@ def main():
|
|
|
387
577
|
)
|
|
388
578
|
|
|
389
579
|
args.output.mkdir(parents=True, exist_ok=True)
|
|
580
|
+
report_path = args.report or (args.output / "retrieval_report.json")
|
|
390
581
|
records = read_doi_file(args.input)
|
|
391
582
|
print(f"Loaded {len(records)} DOIs from {args.input}")
|
|
392
583
|
|
|
393
|
-
stats = {"oa": 0, "pmc": 0, "fail": 0, "skip": 0}
|
|
584
|
+
stats = {"arxiv": 0, "oa": 0, "pmc": 0, "fail": 0, "skip": 0}
|
|
585
|
+
results: dict[str, tuple[str, str]] = {}
|
|
394
586
|
|
|
395
587
|
for i, rec in enumerate(records, 1):
|
|
396
588
|
doi = rec["doi"]
|
|
397
589
|
pmid = rec.get("pmid", "")
|
|
398
590
|
print(f" [{i}/{len(records)}] {doi}", end=" … ", flush=True)
|
|
399
591
|
|
|
400
|
-
status = process_doi(doi, args.output, args.email, pmid)
|
|
592
|
+
status, source = process_doi(doi, args.output, args.email, pmid)
|
|
593
|
+
results[doi] = (status, source)
|
|
401
594
|
stats[status] += 1
|
|
402
595
|
|
|
403
|
-
labels = {"oa": "OK (OA)", "pmc": "OK (PMC)",
|
|
596
|
+
labels = {"arxiv": "OK (arXiv)", "oa": "OK (OA)", "pmc": "OK (PMC)",
|
|
404
597
|
"fail": "FAIL", "skip": "SKIP"}
|
|
405
598
|
print(labels[status])
|
|
406
599
|
time.sleep(0.5)
|
|
407
600
|
|
|
601
|
+
# Best-effort title cross-check on successful downloads (needs pdftotext).
|
|
602
|
+
extracted: dict[str, str] = {}
|
|
603
|
+
have_titles = any(r.get("title") for r in records)
|
|
604
|
+
if have_titles and shutil.which("pdftotext"):
|
|
605
|
+
for rec in records:
|
|
606
|
+
doi = rec["doi"]
|
|
607
|
+
if not rec.get("title"):
|
|
608
|
+
continue
|
|
609
|
+
status, _ = results.get(doi, ("fail", ""))
|
|
610
|
+
if status not in RETRIEVED_STATUSES:
|
|
611
|
+
continue
|
|
612
|
+
path = args.output / f"{safe_doi_name(doi)}.pdf"
|
|
613
|
+
if path.exists():
|
|
614
|
+
text = extract_pdf_text(path)
|
|
615
|
+
if text:
|
|
616
|
+
extracted[doi] = text
|
|
617
|
+
|
|
618
|
+
report = build_report(records, results, args.output, extracted)
|
|
619
|
+
report_path.parent.mkdir(parents=True, exist_ok=True)
|
|
620
|
+
report_path.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
|
|
621
|
+
|
|
408
622
|
print(f"\n--- Summary ---")
|
|
623
|
+
print(f" arXiv: {stats['arxiv']}")
|
|
409
624
|
print(f" OA: {stats['oa']}")
|
|
410
625
|
print(f" PMC: {stats['pmc']}")
|
|
411
626
|
print(f" Failed: {stats['fail']}")
|
|
412
627
|
print(f" Skipped: {stats['skip']}")
|
|
413
|
-
total = stats["oa"] + stats["pmc"] + stats["fail"]
|
|
628
|
+
total = stats["arxiv"] + stats["oa"] + stats["pmc"] + stats["fail"]
|
|
414
629
|
if total > 0:
|
|
415
|
-
pct = (stats["oa"] + stats["pmc"]) / total * 100
|
|
630
|
+
pct = (stats["arxiv"] + stats["oa"] + stats["pmc"]) / total * 100
|
|
416
631
|
print(f" Success: {pct:.0f}%")
|
|
632
|
+
mismatches = report["counts"]["title_mismatch"]
|
|
633
|
+
if mismatches:
|
|
634
|
+
print(f" Title mismatches flagged: {mismatches} (see report)")
|
|
635
|
+
print(f" Report: {report_path}")
|
|
417
636
|
|
|
418
637
|
# Write failed DOIs for manual retrieval
|
|
419
638
|
if stats["fail"] > 0:
|
|
@@ -422,10 +641,10 @@ def main():
|
|
|
422
641
|
f.write("# DOIs needing manual retrieval\n")
|
|
423
642
|
f.write("# Options: institutional access, ILL\n\n")
|
|
424
643
|
for rec in records:
|
|
425
|
-
|
|
426
|
-
pdf = args.output / f"{
|
|
644
|
+
doi = rec["doi"]
|
|
645
|
+
pdf = args.output / f"{safe_doi_name(doi)}.pdf"
|
|
427
646
|
if not existing_pdf_ok(pdf):
|
|
428
|
-
f.write(f"{
|
|
647
|
+
f.write(f"{doi}\n")
|
|
429
648
|
print(f" Manual list: {fail_path}")
|
|
430
649
|
|
|
431
650
|
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"counts": {
|
|
3
|
+
"total": 4,
|
|
4
|
+
"retrieved": 3,
|
|
5
|
+
"not_retrieved": 1,
|
|
6
|
+
"title_mismatch": 1
|
|
7
|
+
},
|
|
8
|
+
"items": [
|
|
9
|
+
{"doi": "10.1111/valid.match", "status": "oa", "source": "unpaywall", "title_match": "match"},
|
|
10
|
+
{"doi": "10.2222/label.mismatch", "status": "oa", "source": "openalex", "title_match": "mismatch"},
|
|
11
|
+
{"doi": "10.3333/no.text", "status": "pmc", "source": "pmc", "title_match": "unavailable"},
|
|
12
|
+
{"doi": "10.9999/not.retrieved", "status": "fail", "source": "", "title_match": "unavailable"}
|
|
13
|
+
]
|
|
14
|
+
}
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
{
|
|
2
|
+
"10.1111/valid.match": "Deep Learning for Pulmonary Nodule Detection on Chest CT\nAbstract: we present a convolutional neural network trained to detect pulmonary nodules.",
|
|
3
|
+
"10.2222/label.mismatch": "Genome-wide association study of type 2 diabetes in an Asian cohort\nIntroduction: we genotyped participants to identify susceptibility loci."
|
|
4
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Offline, network-free challenge for fetch_oa.build_report + title tri-state.
|
|
3
|
+
|
|
4
|
+
Loads committed fixtures (worklist.tsv, results.json, extracted_text.json),
|
|
5
|
+
fabricates stub PDFs in a temp dir for the "retrieved" DOIs, runs
|
|
6
|
+
fetch_oa.build_report, and asserts the resulting projection equals
|
|
7
|
+
expected/projection.json. Exercises read_doi_file (TSV+title), build_report,
|
|
8
|
+
and classify_title_match (match / mismatch / unavailable) without touching the
|
|
9
|
+
network or pdftotext. Stdlib-only.
|
|
10
|
+
"""
|
|
11
|
+
import importlib.util
|
|
12
|
+
import json
|
|
13
|
+
import sys
|
|
14
|
+
import tempfile
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
HERE = Path(__file__).resolve().parent
|
|
18
|
+
ENGINE = HERE.parent / "fetch_oa.py"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def load_engine():
|
|
22
|
+
spec = importlib.util.spec_from_file_location("fetch_oa", ENGINE)
|
|
23
|
+
mod = importlib.util.module_from_spec(spec)
|
|
24
|
+
spec.loader.exec_module(mod)
|
|
25
|
+
return mod
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def projection(report: dict) -> dict:
|
|
29
|
+
items = sorted(
|
|
30
|
+
({"doi": i["doi"], "status": i["status"], "source": i["source"],
|
|
31
|
+
"title_match": i["title_match"]} for i in report["items"]),
|
|
32
|
+
key=lambda x: x["doi"],
|
|
33
|
+
)
|
|
34
|
+
return {"counts": report["counts"], "items": items}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def main() -> int:
|
|
38
|
+
assert ENGINE.exists(), f"ENV-ERR: {ENGINE} missing"
|
|
39
|
+
m = load_engine()
|
|
40
|
+
|
|
41
|
+
records = m.read_doi_file(HERE / "worklist.tsv")
|
|
42
|
+
results = {k: tuple(v) for k, v in
|
|
43
|
+
json.loads((HERE / "results.json").read_text()).items()}
|
|
44
|
+
extracted = json.loads((HERE / "extracted_text.json").read_text())
|
|
45
|
+
expected = json.loads((HERE / "expected" / "projection.json").read_text())
|
|
46
|
+
|
|
47
|
+
fails = []
|
|
48
|
+
|
|
49
|
+
def check(label, cond):
|
|
50
|
+
print(f" {'PASS' if cond else 'FAIL'} {label}")
|
|
51
|
+
if not cond:
|
|
52
|
+
fails.append(label)
|
|
53
|
+
|
|
54
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
55
|
+
outdir = Path(tmp)
|
|
56
|
+
for rec in records:
|
|
57
|
+
status, _ = results.get(rec["doi"], ("fail", ""))
|
|
58
|
+
if status in m.RETRIEVED_STATUSES:
|
|
59
|
+
stub = outdir / f"{m.safe_doi_name(rec['doi'])}.pdf"
|
|
60
|
+
stub.write_bytes(b"%PDF-1.4\n" + b"0" * (11 * 1024))
|
|
61
|
+
report = m.build_report(records, results, outdir, extracted)
|
|
62
|
+
proj = projection(report)
|
|
63
|
+
|
|
64
|
+
by_doi = {i["doi"]: i for i in proj["items"]}
|
|
65
|
+
check("valid -> title_match 'match'",
|
|
66
|
+
by_doi["10.1111/valid.match"]["title_match"] == "match")
|
|
67
|
+
check("mislabel -> title_match 'mismatch'",
|
|
68
|
+
by_doi["10.2222/label.mismatch"]["title_match"] == "mismatch")
|
|
69
|
+
check("no-text -> title_match 'unavailable'",
|
|
70
|
+
by_doi["10.3333/no.text"]["title_match"] == "unavailable")
|
|
71
|
+
check("missing -> status 'fail'",
|
|
72
|
+
by_doi["10.9999/not.retrieved"]["status"] == "fail")
|
|
73
|
+
check("counts.retrieved == 3", proj["counts"]["retrieved"] == 3)
|
|
74
|
+
check("counts.not_retrieved == 1", proj["counts"]["not_retrieved"] == 1)
|
|
75
|
+
check("counts.title_mismatch == 1", proj["counts"]["title_mismatch"] == 1)
|
|
76
|
+
check("projection matches expected/projection.json", proj == expected)
|
|
77
|
+
|
|
78
|
+
if fails:
|
|
79
|
+
print(f"FAILURES: {len(fails)}")
|
|
80
|
+
return 1
|
|
81
|
+
print("ALL PASS")
|
|
82
|
+
return 0
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
if __name__ == "__main__":
|
|
86
|
+
sys.exit(main())
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Network-free challenge: fetch_oa report builder + title-match tri-state.
|
|
3
|
+
# Mirrors CI usage: `bash skills/fulltext-retrieval/fetch_oa_report_challenge/verify.sh`
|
|
4
|
+
set -euo pipefail
|
|
5
|
+
DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
6
|
+
python3 "$DIR/run_challenge.py"
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
DOI PMID Title
|
|
2
|
+
10.1111/valid.match 1001 Deep learning for pulmonary nodule detection on chest CT
|
|
3
|
+
10.2222/label.mismatch 1002 Automated segmentation of cardiac MRI using transformers
|
|
4
|
+
10.3333/no.text 1003 Federated learning for multi-center radiology models
|
|
5
|
+
10.9999/not.retrieved 1004 A paywalled study with no open-access copy
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Find Available PDF — batch trigger for Zotero (user-run snippet)
|
|
3
|
+
* ----------------------------------------------------------------
|
|
4
|
+
* Attaches full-text PDFs to library items using Zotero's OWN
|
|
5
|
+
* "Find Available PDF" resolver. This reuses whatever OpenURL resolver,
|
|
6
|
+
* institutional proxy, or library configuration YOU have already set in
|
|
7
|
+
* Zotero — so it typically retrieves far more than open-access-only resolvers,
|
|
8
|
+
* yet no credentials, proxy hosts, or institutional identifiers ever leave
|
|
9
|
+
* your Zotero client. Nothing here is hard-coded to any institution.
|
|
10
|
+
*
|
|
11
|
+
* HOW TO RUN
|
|
12
|
+
* 1. In Zotero, select the items (or open the collection) you want PDFs for.
|
|
13
|
+
* 2. Tools → Developer → Run JavaScript.
|
|
14
|
+
* 3. Paste this whole snippet and click Run.
|
|
15
|
+
* 4. The result panel prints a JSON summary: considered / attached /
|
|
16
|
+
* alreadyHadPDF / missing (DOIs still without a PDF).
|
|
17
|
+
*
|
|
18
|
+
* NO-CODE FALLBACK
|
|
19
|
+
* Select items → right-click → "Find Available PDF" does the same thing
|
|
20
|
+
* interactively. Use it if you prefer not to run a script.
|
|
21
|
+
*
|
|
22
|
+
* VERSION NOTE
|
|
23
|
+
* Zotero 7 exposes a batch Zotero.Attachments.addAvailablePDFs(items);
|
|
24
|
+
* Zotero 6 only has the per-item Zotero.Attachments.addAvailablePDF(item).
|
|
25
|
+
* This snippet prefers the batch call and falls back to per-item.
|
|
26
|
+
*
|
|
27
|
+
* NOTE: results are user-initiated and depend on your live Zotero session;
|
|
28
|
+
* they are NOT reproducible CI evidence. Record retrieved/not-retrieved
|
|
29
|
+
* outcomes from the printed summary into your retrieval report manually.
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
var pane = Zotero.getActiveZoteroPane();
|
|
33
|
+
var items = pane.getSelectedItems().filter(function (it) { return it.isRegularItem(); });
|
|
34
|
+
|
|
35
|
+
// Fall back to the whole selected collection if nothing is selected.
|
|
36
|
+
if (!items.length) {
|
|
37
|
+
var collection = pane.getSelectedCollection();
|
|
38
|
+
if (collection) {
|
|
39
|
+
items = collection.getChildItems().filter(function (it) { return it.isRegularItem(); });
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function hasPDF(item) {
|
|
44
|
+
return item.getAttachments().some(function (id) {
|
|
45
|
+
var att = Zotero.Items.get(id);
|
|
46
|
+
if (!att) return false;
|
|
47
|
+
if (typeof att.isPDFAttachment === "function") return att.isPDFAttachment();
|
|
48
|
+
return att.attachmentContentType === "application/pdf";
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
var todo = items.filter(function (it) { return !hasPDF(it); });
|
|
53
|
+
var alreadyHadPDF = items.length - todo.length;
|
|
54
|
+
|
|
55
|
+
if (typeof Zotero.Attachments.addAvailablePDFs === "function") {
|
|
56
|
+
await Zotero.Attachments.addAvailablePDFs(todo); // Zotero 7 batch
|
|
57
|
+
} else {
|
|
58
|
+
for (let it of todo) { // Zotero 6 per-item
|
|
59
|
+
try {
|
|
60
|
+
await Zotero.Attachments.addAvailablePDF(it);
|
|
61
|
+
} catch (e) {
|
|
62
|
+
Zotero.debug("addAvailablePDF failed for item " + it.id + ": " + e);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
var attached = 0;
|
|
68
|
+
var missing = [];
|
|
69
|
+
for (let it of todo) {
|
|
70
|
+
if (hasPDF(it)) {
|
|
71
|
+
attached++;
|
|
72
|
+
} else {
|
|
73
|
+
missing.push(it.getField("DOI") || it.getField("title") || ("itemID:" + it.id));
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
return JSON.stringify({
|
|
78
|
+
considered: items.length,
|
|
79
|
+
attached: attached,
|
|
80
|
+
alreadyHadPDF: alreadyHadPDF,
|
|
81
|
+
missing: missing
|
|
82
|
+
}, null, 2);
|
|
@@ -8,12 +8,14 @@ when_to_use: "Batch-download open-access full-text PDFs from a DOI list; optiona
|
|
|
8
8
|
when_NOT_to_use: "Finding or verifying citations (use search-lit / verify-refs). Retrieving paywalled or non-open-access content."
|
|
9
9
|
|
|
10
10
|
inputs:
|
|
11
|
-
- path: "DOI
|
|
11
|
+
- path: "DOI worklist (.txt one-per-line, or .tsv/.csv/.md with a DOI column; optional PMID, Title)"
|
|
12
12
|
schema: csv
|
|
13
13
|
required: true
|
|
14
14
|
outputs:
|
|
15
15
|
- path: "downloaded open-access PDFs (pdfs/)"
|
|
16
|
+
- path: "pdfs/retrieval_report.json (per-DOI status/source/title_match tri-state)"
|
|
16
17
|
- path: "optional PDF-to-Markdown conversions"
|
|
18
|
+
- path: "references/find_available_pdf.js (user-run Zotero 'Find Available PDF' batch snippet)"
|
|
17
19
|
|
|
18
20
|
deterministic_scripts:
|
|
19
21
|
- fetch_oa.py
|
|
@@ -35,8 +37,11 @@ safety_boundaries:
|
|
|
35
37
|
- "Validates each download (>=10 KB and a %PDF- header) before accepting it."
|
|
36
38
|
known_limitations:
|
|
37
39
|
- "Only open-access content is retrievable; non-OA DOIs fail by design rather than fetching from unauthorized sources."
|
|
40
|
+
- "Higher-yield in-library retrieval (find_available_pdf.js) is user-initiated inside Zotero and uses the user's own proxy/OpenURL config; it is not reproducible CI evidence."
|
|
41
|
+
- "Title cross-check is best-effort: it needs a Title column plus pdftotext (poppler); otherwise title_match is 'unavailable'. A mismatch is flagged, never auto-rejected."
|
|
38
42
|
- "PDF-to-Markdown conversion requires the optional pymupdf4llm dependency (AGPL-3.0 or commercial license)."
|
|
39
43
|
validation_commands:
|
|
40
|
-
- "
|
|
44
|
+
- "bash fetch_oa_report_challenge/verify.sh # offline report-builder + title tri-state (CI-wired)"
|
|
45
|
+
- "python fetch_oa.py dois.txt -o pdfs/ -e <email> --report pdfs/retrieval_report.json --verbose # per-DOI source trace"
|
|
41
46
|
- "verify each output begins with %PDF- and is at least 10 KB"
|
|
42
47
|
evidence_surface: bundled_script
|
package/skills/lit-sync/SKILL.md
CHANGED
|
@@ -65,6 +65,9 @@ Direct hand edits to `refs.bib` are drift — revert on sight.
|
|
|
65
65
|
▼ Phase 2.5: refs.bib snapshot refresh
|
|
66
66
|
Trigger Better BibTeX auto-export → verify manuscript/_src/refs.bib mtime updated
|
|
67
67
|
│
|
|
68
|
+
▼ Phase 2.7: Fulltext Retrieval (opt-in)
|
|
69
|
+
Disk OA PDFs via /fulltext-retrieval + in-library via find_available_pdf.js → reconcile report
|
|
70
|
+
│
|
|
68
71
|
▼ Phase 3: Obsidian Literature Notes
|
|
69
72
|
Create Literature/{citekey}.md (empty note OK — fill later with highlights)
|
|
70
73
|
│
|
|
@@ -116,8 +119,16 @@ collection key for future use.
|
|
|
116
119
|
For each entry:
|
|
117
120
|
|
|
118
121
|
1. Use `zotero_search_items` to search by DOI or title — if already present, skip.
|
|
122
|
+
This search-first step is what prevents duplicates; `zotero_add_by_doi` does **not**
|
|
123
|
+
dedupe by itself (it fetches CrossRef and creates the item), so never skip the search.
|
|
119
124
|
2. Otherwise call `zotero_add_by_doi` (when a DOI is available) or
|
|
120
125
|
`zotero_add_by_url` (falling back to the PubMed URL when no DOI is available).
|
|
126
|
+
- `zotero_add_by_doi` accepts an `attach_mode` argument that governs the **OA child-PDF
|
|
127
|
+
attach attempt at add time** (the installed server treats `linked_url` as "bookmark the
|
|
128
|
+
PDF URL"; other values download/import). Set it when you want a PDF attached during the
|
|
129
|
+
add. Exact accepted values are server-version-specific — verify against the connected
|
|
130
|
+
server. Do **not** use `zotero_add_from_file` to attach a PDF to an item added here: it
|
|
131
|
+
has no parent-item argument and would create a duplicate parent item.
|
|
121
132
|
3. Use `zotero_manage_collections` to place the item in the project collection.
|
|
122
133
|
|
|
123
134
|
### Step 2.3: Result report
|
|
@@ -214,6 +225,62 @@ If refresh failed, set `refs_bib_refreshed: false` and include `reason`. `/verif
|
|
|
214
225
|
|
|
215
226
|
---
|
|
216
227
|
|
|
228
|
+
## Phase 2.7: Fulltext Retrieval (opt-in, owner-only)
|
|
229
|
+
|
|
230
|
+
**Run only when the user asks for full text** (e.g. "download the PDFs", "fetch full
|
|
231
|
+
text", or a worklist supplied with that intent). Default `/lit-sync` stays metadata-only
|
|
232
|
+
and network-light — do not auto-run this phase. Runs after items are in Zotero (Phase 2)
|
|
233
|
+
and the snapshot is verified (Phase 2.5), before Obsidian notes (Phase 3).
|
|
234
|
+
|
|
235
|
+
There are two complementary retrieval routes; offer both and reconcile them in one report:
|
|
236
|
+
|
|
237
|
+
### Route A — disk OA PDFs (for downstream skills)
|
|
238
|
+
|
|
239
|
+
Delegate to the `/fulltext-retrieval` engine (do **not** re-implement the OA cascade or
|
|
240
|
+
import its code; invoke it by path). Resolve the engine as:
|
|
241
|
+
|
|
242
|
+
```bash
|
|
243
|
+
ENGINE="${MEDSCI_SKILLS_ROOT:-$HOME/workspace/medsci-skills}/skills/fulltext-retrieval/fetch_oa.py"
|
|
244
|
+
python3 "$ENGINE" <worklist> -o pdfs/ -e <contact-email> --report pdfs/retrieval_report.json
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
`<worklist>` is the DOI/PMID(/Title) list — the Phase-1 `.bib` DOIs, the worklist supplied
|
|
248
|
+
in the standalone mode below, or the project collection's DOIs. Output: `pdfs/*.pdf` for
|
|
249
|
+
`/meta-analysis`, `/obsidian-paper-vault`, and `pdf_to_md.py`, plus
|
|
250
|
+
`pdfs/retrieval_report.json` (per-DOI `status`/`source`/`title_match`).
|
|
251
|
+
|
|
252
|
+
### Route B — in-library PDFs (Zotero-native, higher yield, proxy-aware)
|
|
253
|
+
|
|
254
|
+
Emit `${MEDSCI_SKILLS_ROOT:-$HOME/workspace/medsci-skills}/skills/fulltext-retrieval/references/find_available_pdf.js`
|
|
255
|
+
for the user to paste into Zotero (*Tools → Developer → Run JavaScript*) with the project
|
|
256
|
+
collection selected. It triggers Zotero's own `addAvailablePDF`/`addAvailablePDFs`, which
|
|
257
|
+
reuse the **user's** OpenURL resolver / institutional proxy — so it typically retrieves more
|
|
258
|
+
than OA-only, while **no credentials or institutional identifiers enter this skill**. The
|
|
259
|
+
no-code equivalent is right-click → "Find Available PDF". This route is user-initiated and
|
|
260
|
+
session-dependent; record its `{attached, missing}` summary from the printed JSON.
|
|
261
|
+
|
|
262
|
+
### Report
|
|
263
|
+
|
|
264
|
+
Merge Route A's `pdfs/retrieval_report.json` (and the user-reported Route B summary) into
|
|
265
|
+
`references/fulltext_retrieval.json` (owner of this file is `/lit-sync`):
|
|
266
|
+
|
|
267
|
+
```json
|
|
268
|
+
{
|
|
269
|
+
"schema_version": 1,
|
|
270
|
+
"retrieved_oa_disk": [{"doi": "...", "source": "unpaywall", "file": "...", "title_match": "match"}],
|
|
271
|
+
"retrieved_zotero_native": [{"doi": "...", "via": "addAvailablePDF"}],
|
|
272
|
+
"not_retrieved": [{"doi": "...", "journal": "..."}],
|
|
273
|
+
"institutional_fallback": ["<DOIs needing institutional access / ILL / author contact>"],
|
|
274
|
+
"title_mismatch_flagged": ["<DOIs whose downloaded PDF title did not match>"]
|
|
275
|
+
}
|
|
276
|
+
```
|
|
277
|
+
|
|
278
|
+
Also append a short `fulltext` block (counts) to `references/zotero_collection.json`.
|
|
279
|
+
`not_retrieved` DOIs are candidates for institutional access, interlibrary loan, or author
|
|
280
|
+
contact — never bypass paywalls or access controls from this skill.
|
|
281
|
+
|
|
282
|
+
---
|
|
283
|
+
|
|
217
284
|
## Phase 3: Obsidian Literature Notes
|
|
218
285
|
|
|
219
286
|
### Step 3.1: Check existing literature notes
|
|
@@ -414,16 +481,23 @@ curl -s "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi?db=pubmed&i
|
|
|
414
481
|
| jq -r '.result | to_entries[] | select(.key != "uids") | "\(.value.uid)\t\(.value.elocationid)\t\(.value.title)"'
|
|
415
482
|
```
|
|
416
483
|
|
|
417
|
-
For each resolved DOI
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
`references/zotero_collection.json`.
|
|
484
|
+
For each resolved DOI, search-first with `zotero_search_items`, then call
|
|
485
|
+
`zotero_add_by_doi` — the search is what dedupes (add-by-doi alone does not). For items
|
|
486
|
+
already in the library (detected via `zotero_search_items` by DOI), use
|
|
487
|
+
`zotero_manage_collections` to attach them to the project collection **without re-adding** —
|
|
488
|
+
re-adding by URL/PubMed-URL would bypass the search dedup and create duplicates. Record both
|
|
489
|
+
`added` and `existing` items in `references/zotero_collection.json`.
|
|
423
490
|
|
|
424
491
|
If a PMID has no DOI in PubMed (rare; older papers, non-indexed), fall back to
|
|
425
492
|
`zotero_add_by_url` with the PubMed URL and mark the entry as `no_doi: true`.
|
|
426
493
|
|
|
494
|
+
### Worklist ingestion (DOI/PMID/Title; no .bib)
|
|
495
|
+
When the user supplies a worklist file (a `.tsv`/`.csv`/`.md` table with a `DOI` column,
|
|
496
|
+
optional `PMID`/`Title`, or a plain DOI-per-line list — e.g. an SR include set), enter
|
|
497
|
+
Phase 2 directly from it: resolve any PMID-only rows to DOIs (esummary above), then run the
|
|
498
|
+
search-first dedupe + add loop. The same worklist file feeds Phase 2.7 Route A
|
|
499
|
+
(`fetch_oa.py` reads `.tsv`/`.csv`/`.md`/plain natively), so no reformatting is needed.
|
|
500
|
+
|
|
427
501
|
---
|
|
428
502
|
|
|
429
503
|
## Safety Rules
|
|
@@ -439,6 +513,12 @@ If a PMID has no DOI in PubMed (rare; older papers, non-indexed), fall back to
|
|
|
439
513
|
collection is created.
|
|
440
514
|
6. **Never write `refs.bib` directly.** Only Better BibTeX auto-export may write that file. If auto-export is broken, fix the Zotero setup rather than writing the file from this skill.
|
|
441
515
|
7. **Owner-only execution.** If the current user is a collaborator (no Zotero access per `SSOT.yaml` `reference_manager.required_for`), abort with instructions to flag `[@NEW:topic]` placeholders in the manuscript and notify the owner.
|
|
516
|
+
8. **Fulltext boundary (Phase 2.7).** Retrieve full text only via OA APIs (the
|
|
517
|
+
`/fulltext-retrieval` engine) and the user-run Zotero "Find Available PDF" snippet
|
|
518
|
+
(which uses the user's own proxy config). Never automate authenticated browser sessions,
|
|
519
|
+
never bypass paywalls/access controls, and never hard-code institutional proxies,
|
|
520
|
+
credentials, or hosts into this skill. `not_retrieved` items are routed to institutional
|
|
521
|
+
access / ILL / author contact, not worked around.
|
|
442
522
|
|
|
443
523
|
## Anti-Hallucination
|
|
444
524
|
|
|
@@ -8,6 +8,7 @@ when_to_use:
|
|
|
8
8
|
- After /search-lit when verified candidates need to flow into Zotero + refs.bib
|
|
9
9
|
- Cross-cutting concept-note extraction from accumulated literature into Obsidian
|
|
10
10
|
- Refreshing manuscript/_src/refs.bib via Better BibTeX auto-export before a build
|
|
11
|
+
- Opt-in full-text retrieval (Phase 2.7) — disk OA PDFs via /fulltext-retrieval + in-library "Find Available PDF"
|
|
11
12
|
when_NOT_to_use:
|
|
12
13
|
- Pure literature search without sync (use /search-lit)
|
|
13
14
|
- Hand-editing refs.bib (BBT is the sole writer; never touch the file directly)
|
|
@@ -17,10 +18,11 @@ inputs:
|
|
|
17
18
|
- PMID list, DOI list, or Zotero collection name
|
|
18
19
|
outputs:
|
|
19
20
|
- references/zotero_collection.json
|
|
21
|
+
- references/fulltext_retrieval.json # SOLE WRITER; opt-in Phase 2.7 retrieval report
|
|
20
22
|
- manuscript/_src/refs.bib # SOLE WRITER via Better BibTeX auto-export "Keep updated"
|
|
21
23
|
- obsidian_literature_notes
|
|
22
24
|
deterministic_scripts:
|
|
23
|
-
- none_required # leverages Zotero MCP + Better BibTeX auto-export GUI
|
|
25
|
+
- none_required # leverages Zotero MCP + Better BibTeX auto-export GUI; Phase 2.7 invokes /fulltext-retrieval fetch_oa.py
|
|
24
26
|
side_effects:
|
|
25
27
|
- may_update_zotero
|
|
26
28
|
- may_write_obsidian_notes
|
|
@@ -263,9 +263,9 @@ Total: 13 references
|
|
|
263
263
|
|
|
264
264
|
If a Zotero MCP server is available, integrate search results with the user's library:
|
|
265
265
|
|
|
266
|
-
1. **
|
|
267
|
-
2. **
|
|
268
|
-
3. **
|
|
266
|
+
1. **Check for duplicates first**: Use `zotero_search_items` (by DOI) to skip papers already in the library — this search-first step is what dedupes; `zotero_add_by_doi` does not dedupe on its own.
|
|
267
|
+
2. **Add papers to Zotero**: Use `zotero_add_by_doi` for DOI-based import (its `attach_mode` argument governs the OA PDF attach attempt at add time).
|
|
268
|
+
3. **Organize into collections**: Use `zotero_manage_collections` to file into the relevant project collection.
|
|
269
269
|
4. **Leverage annotations**: Use `zotero_get_annotations` to reference the user's prior reading notes.
|
|
270
270
|
5. **Write sync audit**: Record collection key, added/skipped/failed counts, and
|
|
271
271
|
unsynced entries in `references/zotero_collection.json` so Zotero status is
|
|
@@ -277,77 +277,32 @@ If a Zotero MCP server is available, integrate search results with the user's li
|
|
|
277
277
|
|
|
278
278
|
### Phase 5: Full-Text Retrieval
|
|
279
279
|
|
|
280
|
-
|
|
281
|
-
|
|
280
|
+
Full-text PDF retrieval is **delegated to `/fulltext-retrieval`** — the single authored
|
|
281
|
+
home of the open-access cascade (arXiv → Unpaywall → PMC → OpenAlex → Crossref → landing
|
|
282
|
+
page, each validated with a `%PDF-` header + ≥10 KB size). Do **not** re-implement OA
|
|
283
|
+
fetching here.
|
|
282
284
|
|
|
283
|
-
|
|
285
|
+
Pass the verified candidate DOIs from `references/library.bib`:
|
|
284
286
|
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
email = os.environ.get("UNPAYWALL_EMAIL", "user@example.com")
|
|
291
|
-
url = f"https://api.unpaywall.org/v2/{doi}?email={email}"
|
|
292
|
-
r = requests.get(url).json()
|
|
293
|
-
if r.get("best_oa_location", {}).get("url_for_pdf"):
|
|
294
|
-
pdf_url = r["best_oa_location"]["url_for_pdf"]
|
|
295
|
-
```
|
|
296
|
-
|
|
297
|
-
2. **PubMed Central (PMC)**:
|
|
298
|
-
- Convert PMID to PMCID via NCBI ID Converter
|
|
299
|
-
- Download from PMC OA service: `https://www.ncbi.nlm.nih.gov/pmc/articles/PMC{id}/pdf/`
|
|
300
|
-
|
|
301
|
-
3. **OpenAlex API** (additional OA discovery):
|
|
302
|
-
```python
|
|
303
|
-
url = f"https://api.openalex.org/works/https://doi.org/{doi}"
|
|
304
|
-
# Requires polite pool: add email in User-Agent header or mailto= param
|
|
305
|
-
r = requests.get(url, headers={"User-Agent": f"MyApp/1.0 (mailto:{email})"}).json()
|
|
306
|
-
oa_url = r.get("open_access", {}).get("oa_url")
|
|
307
|
-
```
|
|
308
|
-
|
|
309
|
-
4. **CrossRef landing page**: Follow `https://api.crossref.org/works/{doi}` → publisher link
|
|
310
|
-
→ scrape `<meta name="citation_pdf_url">` tag
|
|
311
|
-
|
|
312
|
-
#### Phase 5b: Alternative Sources
|
|
313
|
-
|
|
314
|
-
Some researchers use alternative access methods for paywalled content.
|
|
315
|
-
**Users are responsible for ensuring compliance with their institutional access policies.**
|
|
316
|
-
|
|
317
|
-
If an environment variable (e.g., `SCIHUB_BASE`) is set, the skill may use it as an
|
|
318
|
-
alternative PDF source. No specific URLs are provided here — users configure this themselves.
|
|
319
|
-
|
|
320
|
-
Other options:
|
|
321
|
-
- **Institutional proxy/VPN**: Access publisher sites through institutional EZproxy or VPN
|
|
322
|
-
- **Interlibrary loan (ILL)**: Request through library services for papers not otherwise available
|
|
323
|
-
- **Author contact**: Email corresponding authors for preprints
|
|
324
|
-
|
|
325
|
-
#### PDF Validation
|
|
287
|
+
```bash
|
|
288
|
+
ENGINE="${MEDSCI_SKILLS_ROOT:-$HOME/workspace/medsci-skills}/skills/fulltext-retrieval/fetch_oa.py"
|
|
289
|
+
# extract DOIs from references/library.bib → dois.txt (one per line)
|
|
290
|
+
python3 "$ENGINE" dois.txt -o pdfs/ -e <contact-email> --report pdfs/retrieval_report.json
|
|
291
|
+
```
|
|
326
292
|
|
|
327
|
-
|
|
293
|
+
For Zotero-resident PDFs and higher-yield, proxy-aware retrieval, use `/lit-sync` Phase 2.7,
|
|
294
|
+
which also invokes `/fulltext-retrieval` and triggers Zotero's native "Find Available PDF".
|
|
328
295
|
|
|
329
|
-
|
|
330
|
-
def is_valid_pdf(filepath):
|
|
331
|
-
"""Check that a downloaded file is actually a PDF, not an HTML redirect."""
|
|
332
|
-
import os
|
|
333
|
-
if os.path.getsize(filepath) < 10240: # < 10KB is likely a stub/redirect
|
|
334
|
-
return False
|
|
335
|
-
with open(filepath, 'rb') as f:
|
|
336
|
-
header = f.read(5)
|
|
337
|
-
return header == b'%PDF-'
|
|
338
|
-
```
|
|
296
|
+
#### Alternative sources (legitimate only)
|
|
339
297
|
|
|
340
|
-
|
|
341
|
-
- Verify HTTP `Content-Type: application/pdf` header before saving
|
|
342
|
-
- Files under 10KB are almost always HTML login/redirect pages, not real PDFs
|
|
343
|
-
- Some publishers return CAPTCHA pages — these fail the `%PDF-` check
|
|
298
|
+
For DOIs that open access cannot reach (listed in `pdfs/manual_needed.txt`):
|
|
344
299
|
|
|
345
|
-
|
|
300
|
+
- **Institutional access / proxy / VPN** — through your library's own subscriptions.
|
|
301
|
+
- **Interlibrary loan (ILL)** — request via library services.
|
|
302
|
+
- **Author contact** — email the corresponding author for a copy or preprint.
|
|
346
303
|
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
- NCBI/PMC: 3 requests/sec without API key, 10/sec with `NCBI_API_KEY`
|
|
350
|
-
- General: 2-second minimum interval between requests to any single host
|
|
304
|
+
Never bypass paywalls or publisher access controls, and do not configure unauthorized
|
|
305
|
+
PDF mirrors. Rate limits and PDF validation are handled inside `/fulltext-retrieval`.
|
|
351
306
|
|
|
352
307
|
### Phase 6: Gap Analysis
|
|
353
308
|
|