medsci-skills 5.0.0 → 5.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1488,23 +1488,58 @@
1488
1488
  },
1489
1489
  {
1490
1490
  "path": "skills/fulltext-retrieval/SKILL.md",
1491
- "size": 5830,
1492
- "sha256": "0bdf94398af3d9f144b0b58fbce643d05f5486329472681a39090f135d3ec63f"
1491
+ "size": 8248,
1492
+ "sha256": "d97c19a64d92f66e3c8a1445d893c0f55364f21dc99404fb6f012a3566556e73"
1493
1493
  },
1494
1494
  {
1495
1495
  "path": "skills/fulltext-retrieval/fetch_oa.py",
1496
- "size": 15882,
1497
- "sha256": "b33feb91c7258d1986fbb6168a044b2abe21f3ba1cacadd3a52a38689477f86c"
1496
+ "size": 24503,
1497
+ "sha256": "768f95a30737e67b7f9edee72fe8815a795c9f83451b46551bbe25fc36094f4b"
1498
+ },
1499
+ {
1500
+ "path": "skills/fulltext-retrieval/fetch_oa_report_challenge/expected/projection.json",
1501
+ "size": 522,
1502
+ "sha256": "f24bdb507f66932c0429acefc171abf5221bd644cacdeba5cd7b5c07670b04aa"
1503
+ },
1504
+ {
1505
+ "path": "skills/fulltext-retrieval/fetch_oa_report_challenge/extracted_text.json",
1506
+ "size": 351,
1507
+ "sha256": "9b512bbea3780727e79355dd4b3689be182f086d4afb4184737524449fc44e4d"
1508
+ },
1509
+ {
1510
+ "path": "skills/fulltext-retrieval/fetch_oa_report_challenge/results.json",
1511
+ "size": 175,
1512
+ "sha256": "e46d52e6a9c038ebd5cb8ce89a5961f32c8365e80d189d3ee80bed021a60296a"
1513
+ },
1514
+ {
1515
+ "path": "skills/fulltext-retrieval/fetch_oa_report_challenge/run_challenge.py",
1516
+ "size": 3138,
1517
+ "sha256": "a3c194bb31f319bdd011d4436e5de22cfcc79b7979fbcba4883ed370e4acbeb4"
1518
+ },
1519
+ {
1520
+ "path": "skills/fulltext-retrieval/fetch_oa_report_challenge/verify.sh",
1521
+ "size": 286,
1522
+ "sha256": "f96687fa9d1b213734810dcdfa8126a2efd1af106f5d1cdeb43d58b431ab915b"
1523
+ },
1524
+ {
1525
+ "path": "skills/fulltext-retrieval/fetch_oa_report_challenge/worklist.tsv",
1526
+ "size": 326,
1527
+ "sha256": "d1c4c906052be198275155ee3e9ecf29b04ec8bdfeef97816ca15aa8cefa61fa"
1498
1528
  },
1499
1529
  {
1500
1530
  "path": "skills/fulltext-retrieval/pdf_to_md.py",
1501
1531
  "size": 5398,
1502
1532
  "sha256": "16ba8c61db254b4b356d85686946351daa5ea650d70dca2fd0d5677851c6df4d"
1503
1533
  },
1534
+ {
1535
+ "path": "skills/fulltext-retrieval/references/find_available_pdf.js",
1536
+ "size": 3057,
1537
+ "sha256": "04dc6d13d0bb7c43679e5386f1788edee6b187b2503cd9e250ae4ea63e11cec0"
1538
+ },
1504
1539
  {
1505
1540
  "path": "skills/fulltext-retrieval/skill.yml",
1506
- "size": 1712,
1507
- "sha256": "43f84eeeeedb4676fb80fdca2ac0a84f6a90901b1d4e1d88359a532bb4ead835"
1541
+ "size": 2420,
1542
+ "sha256": "7a457e5f5fde09f57a8ef9d2207dbaab241e56a9bca59ba577d26351bf225346"
1508
1543
  },
1509
1544
  {
1510
1545
  "path": "skills/generate-codebook/SKILL.md",
@@ -1563,8 +1598,8 @@
1563
1598
  },
1564
1599
  {
1565
1600
  "path": "skills/lit-sync/SKILL.md",
1566
- "size": 16378,
1567
- "sha256": "e59b10d8dfdb52318a212e42d403a4372435697880e45643d3864f1374c7e82a"
1601
+ "size": 21126,
1602
+ "sha256": "4ba2d2b6dd4704803c0922993401b837d4135ac1d655d6064229315537a8e818"
1568
1603
  },
1569
1604
  {
1570
1605
  "path": "skills/lit-sync/references/locale/ko/note_templates.md",
@@ -1573,8 +1608,8 @@
1573
1608
  },
1574
1609
  {
1575
1610
  "path": "skills/lit-sync/skill.yml",
1576
- "size": 2522,
1577
- "sha256": "5f656fae91ba46d54f9d619785607bfb57773abc6a872207435ecf7a7315816b"
1611
+ "size": 2780,
1612
+ "sha256": "a04d160e9fc6382c783940b737c9d9348db45e34ea80c69ef905be0a3c357e34"
1578
1613
  },
1579
1614
  {
1580
1615
  "path": "skills/ma-scout/SKILL.md",
@@ -3238,8 +3273,8 @@
3238
3273
  },
3239
3274
  {
3240
3275
  "path": "skills/search-lit/SKILL.md",
3241
- "size": 21637,
3242
- "sha256": "35809486f31f1a26288a27ee0b1383760cf97e0287916ace2e9097eef88182e4"
3276
+ "size": 20281,
3277
+ "sha256": "608644e83ffc357f3b43b65cf7c439c9381f5948c51798672929f7a38aea5024"
3243
3278
  },
3244
3279
  {
3245
3280
  "path": "skills/search-lit/references/parse_pubmed.py",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schema_version": 1,
3
- "version": "5.0.0",
3
+ "version": "5.1.0",
4
4
  "owned_skills": [
5
5
  "academic-aio",
6
6
  "add-journal",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "medsci-skills",
3
- "version": "5.0.0",
3
+ "version": "5.1.0",
4
4
  "description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
5
5
  "license": "SEE LICENSE IN LICENSE",
6
6
  "homepage": "https://github.com/Aperivue/medsci-skills#readme",
@@ -13,10 +13,10 @@ Batch download open-access full-text PDFs from a DOI list using legitimate OA AP
13
13
  ## Pipeline
14
14
 
15
15
  ```
16
- DOI list → Unpaywall → PMC (Europe PMC / OA FTP / web) → OpenAlex → Crossref → landing page
16
+ DOI arXiv (10.48550/arXiv.* DOIs) → Unpaywall → PMC (Europe PMC / OA FTP / web) → OpenAlex → Crossref → landing page
17
17
  ```
18
18
 
19
- Each DOI goes through these sources in order until a valid PDF (≥10 KB, `%PDF-` header) is found.
19
+ Each DOI goes through these sources in order until a valid PDF (≥10 KB, `%PDF-` header) is found. arXiv DOIs (`10.48550/arXiv.2401.01234`, version suffixes, old-style `hep-th/9901001`, or a bare `arXiv:` id) resolve directly to the arXiv PDF first.
20
20
 
21
21
  ## Quick Start
22
22
 
@@ -43,13 +43,20 @@ python fetch_oa.py dois.txt -o pdfs/ -e your@email.com --verbose
43
43
  10.1002/mp.12524
44
44
  ```
45
45
 
46
- **TSV with header** — must contain a `DOI` column, optional `PMID` column:
46
+ **TSV / CSV with header** — must contain a `DOI` column; optional `PMID` and `Title` columns:
47
47
  ```tsv
48
48
  ID Title DOI PMID Year
49
49
  1 Some paper 10.1007/s00330-010-1783-x 20628747 2010
50
50
  ```
51
51
 
52
- When a PMID is available, the PMC lookup is more reliable (PMID → PMCID conversion).
52
+ **Markdown table** a pipe table with a `DOI` column also works:
53
+ ```markdown
54
+ | DOI | PMID | Title |
55
+ |-----|------|-------|
56
+ | 10.1007/s00330-010-1783-x | 20628747 | Some paper |
57
+ ```
58
+
59
+ When a PMID is available, the PMC lookup is more reliable (PMID → PMCID conversion). When a `Title` column is present, downloaded PDFs get a best-effort title cross-check (see *Retrieval report* below).
53
60
 
54
61
  ## PMC Download (JS-Challenge Resistant)
55
62
 
@@ -82,8 +89,48 @@ curl -s "https://www.ncbi.nlm.nih.gov/pmc/utils/idconv/v1.0/?ids=${DOI}&format=j
82
89
  ## Output
83
90
 
84
91
  - PDFs saved as `{DOI_safe}.pdf` (slashes replaced with underscores)
92
+ - `pdfs/retrieval_report.json` — structured per-DOI report (see below)
85
93
  - `manual_needed.txt` — DOIs that could not be retrieved via OA
86
- - Summary with OA/PMC/fail/skip counts
94
+ - Summary with arXiv/OA/PMC/fail/skip counts
95
+
96
+ ## Retrieval report (`--report`)
97
+
98
+ Every run writes a structured report (default `<output>/retrieval_report.json`,
99
+ override with `--report PATH`):
100
+
101
+ ```json
102
+ {
103
+ "schema_version": 1,
104
+ "generated_by": "fetch_oa.py",
105
+ "counts": {"total": 10, "retrieved": 6, "not_retrieved": 4, "title_mismatch": 1},
106
+ "items": [
107
+ {"doi": "10.1007/...", "pmid": "20628747", "title": "...",
108
+ "status": "oa", "source": "unpaywall", "file": "10.1007_....pdf",
109
+ "size_bytes": 482113, "title_match": "match"}
110
+ ]
111
+ }
112
+ ```
113
+
114
+ - `status` ∈ `arxiv | oa | pmc | skip | fail`; `source` names the resolver that succeeded.
115
+ - `title_match` ∈ `match | mismatch | unavailable` (tri-state). It is **best-effort**:
116
+ it needs a `Title` column **and** `pdftotext` (poppler). When either is missing it is
117
+ `unavailable`; a `mismatch` is **flagged** for review and **never** auto-rejects a PDF
118
+ (guards against a publisher serving a wrong/redirect PDF that still passes the `%PDF-` check).
119
+
120
+ ## Attach PDFs into Zotero ("Find Available PDF")
121
+
122
+ OA-only resolvers miss paywalled-but-licensed papers. To attach full text **inside
123
+ Zotero** at a much higher yield, use `references/find_available_pdf.js` — a user-run
124
+ snippet for Zotero's *Tools → Developer → Run JavaScript*. It triggers Zotero's own
125
+ `addAvailablePDF` / `addAvailablePDFs` and therefore reuses **your** OpenURL resolver /
126
+ institutional proxy config; **no credentials, proxy hosts, or institutional identifiers
127
+ are hard-coded or leave your Zotero client**. The no-code equivalent is right-click →
128
+ "Find Available PDF".
129
+
130
+ This path is **user-initiated** and depends on your live Zotero session, so its results
131
+ are recorded manually (not reproducible CI evidence). `/lit-sync` Phase 2.7 orchestrates
132
+ both routes (disk OA via this script + in-library via the snippet) and reconciles them in
133
+ a report.
87
134
 
88
135
  ## Requirements
89
136
 
@@ -2,20 +2,28 @@
2
2
  """
3
3
  Open-access full-text PDF batch retrieval.
4
4
 
5
- Pipeline: Unpaywall → PMC (Europe PMC REST / OA FTP / web)
6
- OpenAlex → Crossref → landing-page scrape.
5
+ Pipeline: arXiv (for 10.48550/arXiv.* DOIs) Unpaywall
6
+ PMC (Europe PMC REST / OA FTP / web) → OpenAlex → Crossref →
7
+ landing-page scrape.
7
8
 
8
9
  Usage:
9
10
  python fetch_oa.py dois.txt --output pdfs/ --email user@example.com
10
- python fetch_oa.py dois.txt -o pdfs/ -e user@example.com --verbose
11
+ python fetch_oa.py worklist.tsv -o pdfs/ -e user@example.com --verbose
12
+ python fetch_oa.py worklist.csv -o pdfs/ -e user@example.com --report pdfs/retrieval_report.json
13
+
14
+ Worklist formats: plain DOI-per-line, or TSV/CSV/Markdown-table with a DOI
15
+ column (optional PMID and Title columns). A Title column enables a best-effort
16
+ title cross-check (via `pdftotext` if installed) that flags mislabeled PDFs.
11
17
  """
12
18
 
13
19
  import argparse
20
+ import csv
21
+ import io
14
22
  import json
15
23
  import logging
16
- import os
17
24
  import re
18
- import sys
25
+ import shutil
26
+ import subprocess
19
27
  import time
20
28
  import urllib.error
21
29
  import urllib.parse
@@ -25,6 +33,9 @@ from pathlib import Path
25
33
 
26
34
  MIN_PDF_BYTES = 10 * 1024
27
35
  USER_AGENT = "medsci-skills/1.0"
36
+ REPORT_SCHEMA_VERSION = 1
37
+ TITLE_MATCH_THRESHOLD = 0.6
38
+ RETRIEVED_STATUSES = ("oa", "pmc", "arxiv", "skip")
28
39
 
29
40
  log = logging.getLogger("fetch_oa")
30
41
 
@@ -38,6 +49,11 @@ def _ua(email: str) -> str:
38
49
  return f"{USER_AGENT} (mailto:{email})"
39
50
 
40
51
 
52
+ def safe_doi_name(doi: str) -> str:
53
+ """Filesystem-safe filename stem for a DOI."""
54
+ return re.sub(r"[^\w\-.]", "_", doi)
55
+
56
+
41
57
  def is_valid_pdf(data: bytes) -> bool:
42
58
  return data.startswith(b"%PDF-") and len(data) >= MIN_PDF_BYTES
43
59
 
@@ -68,6 +84,84 @@ def existing_pdf_ok(path: Path) -> bool:
68
84
  return False
69
85
 
70
86
 
87
+ # ============================================================
88
+ # Title cross-check (pure, offline-testable)
89
+ # ============================================================
90
+
91
+ def normalize_title(text: str) -> str:
92
+ """Lowercase, strip punctuation, collapse whitespace."""
93
+ text = (text or "").lower()
94
+ text = re.sub(r"[^a-z0-9\s]", " ", text)
95
+ return " ".join(text.split())
96
+
97
+
98
+ def title_overlap(expected_title: str, extracted_text: str) -> float:
99
+ """Fraction of meaningful expected-title tokens present in extracted text.
100
+
101
+ Tokens of length <= 2 are dropped as near-stopwords. Returns 0.0 when the
102
+ expected title has no usable tokens.
103
+ """
104
+ expected = {t for t in normalize_title(expected_title).split() if len(t) > 2}
105
+ if not expected:
106
+ return 0.0
107
+ got = set(normalize_title(extracted_text).split())
108
+ return len(expected & got) / len(expected)
109
+
110
+
111
+ def classify_title_match(expected_title: str, extracted_text,
112
+ threshold: float = TITLE_MATCH_THRESHOLD) -> str:
113
+ """Tri-state title check: 'match' | 'mismatch' | 'unavailable'.
114
+
115
+ 'unavailable' when no expected title is supplied or no extracted text is
116
+ available (e.g. pdftotext absent). A low overlap is 'mismatch' — flagged,
117
+ never used to auto-reject a downloaded PDF.
118
+ """
119
+ if not expected_title or not extracted_text:
120
+ return "unavailable"
121
+ return "match" if title_overlap(expected_title, extracted_text) >= threshold else "mismatch"
122
+
123
+
124
+ def extract_pdf_text(path: Path, max_pages: int = 2) -> str | None:
125
+ """Best-effort first-page text via `pdftotext` (poppler). None if unavailable."""
126
+ if not shutil.which("pdftotext"):
127
+ return None
128
+ try:
129
+ out = subprocess.run(
130
+ ["pdftotext", "-f", "1", "-l", str(max_pages), str(path), "-"],
131
+ capture_output=True, timeout=20,
132
+ )
133
+ if out.returncode == 0:
134
+ return out.stdout.decode("utf-8", errors="ignore")
135
+ except (OSError, subprocess.SubprocessError):
136
+ pass
137
+ return None
138
+
139
+
140
+ # ============================================================
141
+ # arXiv (direct, for 10.48550/arXiv.* DOIs)
142
+ # ============================================================
143
+
144
+ _ARXIV_DOI_RE = re.compile(r"^10\.48550/arxiv\.(.+)$", re.IGNORECASE)
145
+ _ARXIV_ID_RE = re.compile(r"^arxiv:(.+)$", re.IGNORECASE)
146
+
147
+
148
+ def arxiv_id_from_doi(doi: str) -> str | None:
149
+ """Extract an arXiv ID from a DataCite arXiv DOI or a bare arXiv: id.
150
+
151
+ Handles new-style (2401.01234, 2401.01234v2) and old-style
152
+ (hep-th/9901001) identifiers; version suffix preserved when present.
153
+ """
154
+ s = (doi or "").strip()
155
+ m = _ARXIV_DOI_RE.match(s) or _ARXIV_ID_RE.match(s)
156
+ return m.group(1).strip() if m else None
157
+
158
+
159
+ def arxiv_pdf_url(doi: str) -> str | None:
160
+ """Direct arXiv PDF URL for an arXiv DOI/ID (None if not an arXiv id)."""
161
+ aid = arxiv_id_from_doi(doi)
162
+ return f"https://arxiv.org/pdf/{aid}" if aid else None
163
+
164
+
71
165
  # ============================================================
72
166
  # 1. Unpaywall
73
167
  # ============================================================
@@ -270,41 +364,34 @@ def download_pdf(url: str, outpath: Path, email: str) -> bool:
270
364
  # 5. Main pipeline
271
365
  # ============================================================
272
366
 
273
- def gather_candidates(doi: str, email: str) -> list[str]:
274
- """Collect OA PDF candidate URLs from multiple sources."""
275
- urls: list[str] = []
276
-
277
- def add(v: str | None):
278
- if v and v not in urls:
279
- urls.append(v)
280
-
281
- add(unpaywall_lookup(doi, email))
282
- for v in openalex_lookup(doi, email):
283
- add(v)
284
- for v in crossref_lookup(doi, email):
285
- add(v)
286
- add(f"https://doi.org/{doi}")
287
- return urls
288
-
289
-
290
367
  def process_doi(doi: str, outdir: Path, email: str,
291
- pmid: str = "") -> str:
292
- """Try to download a PDF for one DOI. Returns status string."""
293
- safe_name = re.sub(r"[^\w\-.]", "_", doi)
294
- outpath = outdir / f"{safe_name}.pdf"
368
+ pmid: str = "") -> tuple[str, str]:
369
+ """Try to download a PDF for one DOI.
370
+
371
+ Returns (status, source):
372
+ status ∈ {"arxiv", "oa", "pmc", "skip", "fail"}
373
+ source identifies the resolver that succeeded (e.g. "unpaywall", "pmc",
374
+ "openalex", "crossref", "landing", "arxiv", "existing", "").
375
+ """
376
+ outpath = outdir / f"{safe_doi_name(doi)}.pdf"
295
377
 
296
378
  if existing_pdf_ok(outpath):
297
- return "skip"
379
+ return ("skip", "existing")
298
380
 
299
381
  # Remove stale stub
300
382
  if outpath.exists():
301
383
  outpath.unlink(missing_ok=True)
302
384
 
385
+ # Step 0: arXiv direct (for 10.48550/arXiv.* DOIs)
386
+ ax_url = arxiv_pdf_url(doi)
387
+ if ax_url and download_pdf(ax_url, outpath, email):
388
+ return ("arxiv", "arxiv")
389
+
303
390
  # Step 1: Unpaywall direct PDF URL (fastest path)
304
391
  uw_url = unpaywall_lookup(doi, email)
305
392
  if uw_url and ".pdf" in uw_url.lower():
306
393
  if download_pdf(uw_url, outpath, email):
307
- return "oa"
394
+ return ("oa", "unpaywall")
308
395
  time.sleep(0.3)
309
396
 
310
397
  # Step 2: PMC (try before slow landing-page scraping)
@@ -312,59 +399,158 @@ def process_doi(doi: str, outdir: Path, email: str,
312
399
  if not pmcid:
313
400
  pmcid = id_to_pmcid(doi, email)
314
401
  if pmcid and download_pmc_pdf(pmcid, outpath, email):
315
- return "pmc"
402
+ return ("pmc", "pmc")
316
403
 
317
404
  # Step 3: OA candidates from OpenAlex, Crossref, landing pages
318
- candidates: list[str] = []
319
- if uw_url and uw_url not in candidates:
320
- candidates.append(uw_url)
405
+ candidates: list[tuple[str, str]] = []
406
+ seen: set[str] = set()
407
+
408
+ def add(source: str, url: str | None):
409
+ if url and url not in seen:
410
+ seen.add(url)
411
+ candidates.append((source, url))
412
+
413
+ add("unpaywall", uw_url)
321
414
  for v in openalex_lookup(doi, email):
322
- if v not in candidates:
323
- candidates.append(v)
415
+ add("openalex", v)
324
416
  for v in crossref_lookup(doi, email):
325
- if v not in candidates:
326
- candidates.append(v)
327
- candidates.append(f"https://doi.org/{doi}")
417
+ add("crossref", v)
418
+ add("landing", f"https://doi.org/{doi}")
328
419
 
329
- for url in candidates:
420
+ for source, url in candidates:
330
421
  if ".pdf" in url.lower():
331
422
  ok = download_pdf(url, outpath, email)
332
423
  else:
333
424
  ok = download_from_landing(url, outpath, email)
334
425
  if ok:
335
- return "oa"
426
+ return ("oa", source)
336
427
  time.sleep(0.3)
337
428
 
338
- return "fail"
429
+ return ("fail", "")
430
+
431
+
432
+ def build_report(records: list[dict], results: dict[str, tuple[str, str]],
433
+ outdir: Path, extracted_text_by_doi: dict[str, str] | None = None,
434
+ threshold: float = TITLE_MATCH_THRESHOLD) -> dict:
435
+ """Assemble a deterministic retrieval report (no network, no I/O writes).
436
+
437
+ records: list of {"doi", "pmid", "title"}.
438
+ results: doi -> (status, source) as returned by process_doi.
439
+ outdir: directory where PDFs were written (used for file/size lookup).
440
+ extracted_text_by_doi: optional doi -> first-page text for title cross-check.
441
+ """
442
+ extracted_text_by_doi = extracted_text_by_doi or {}
443
+ items = []
444
+ for rec in records:
445
+ doi = rec["doi"]
446
+ status, source = results.get(doi, ("fail", ""))
447
+ path = outdir / f"{safe_doi_name(doi)}.pdf"
448
+ have_file = status in RETRIEVED_STATUSES and path.exists()
449
+ size = path.stat().st_size if have_file else 0
450
+ if have_file:
451
+ title_match = classify_title_match(
452
+ rec.get("title", ""), extracted_text_by_doi.get(doi), threshold)
453
+ else:
454
+ title_match = "unavailable"
455
+ items.append({
456
+ "doi": doi,
457
+ "pmid": rec.get("pmid", ""),
458
+ "title": rec.get("title", ""),
459
+ "status": status,
460
+ "source": source,
461
+ "file": path.name if have_file else "",
462
+ "size_bytes": size,
463
+ "title_match": title_match,
464
+ })
465
+
466
+ retrieved = [i for i in items if i["status"] in RETRIEVED_STATUSES]
467
+ not_retrieved = [i for i in items if i["status"] == "fail"]
468
+ return {
469
+ "schema_version": REPORT_SCHEMA_VERSION,
470
+ "generated_by": "fetch_oa.py",
471
+ "counts": {
472
+ "total": len(items),
473
+ "retrieved": len(retrieved),
474
+ "not_retrieved": len(not_retrieved),
475
+ "title_mismatch": sum(1 for i in items if i["title_match"] == "mismatch"),
476
+ },
477
+ "items": items,
478
+ }
479
+
480
+
481
+ def _norm_key(key: str) -> str:
482
+ return (key or "").strip().lstrip("#").strip().lower()
483
+
484
+
485
+ def _records_from_dictrows(rows) -> list[dict]:
486
+ records = []
487
+ for row in rows:
488
+ rec = {"doi": "", "pmid": "", "title": ""}
489
+ for k, v in row.items():
490
+ nk = _norm_key(k)
491
+ if nk in rec:
492
+ rec[nk] = (v or "").strip()
493
+ if rec["doi"]:
494
+ records.append(rec)
495
+ return records
496
+
497
+
498
+ def _records_from_markdown(lines: list[str]) -> list[dict]:
499
+ pipe_rows = [ln for ln in lines if ln.strip().startswith("|")]
500
+ if not pipe_rows:
501
+ return []
502
+
503
+ def cells(line: str) -> list[str]:
504
+ return [c.strip() for c in line.strip().strip("|").split("|")]
505
+
506
+ header = [_norm_key(c) for c in cells(pipe_rows[0])]
507
+ records = []
508
+ for line in pipe_rows[1:]:
509
+ c = cells(line)
510
+ # Skip the |---|---| separator row
511
+ if c and all(set(x) <= set("-: ") for x in c):
512
+ continue
513
+ row = dict(zip(header, c))
514
+ doi = (row.get("doi") or "").strip()
515
+ if doi:
516
+ records.append({
517
+ "doi": doi,
518
+ "pmid": (row.get("pmid") or "").strip(),
519
+ "title": (row.get("title") or "").strip(),
520
+ })
521
+ return records
339
522
 
340
523
 
341
524
  def read_doi_file(path: Path) -> list[dict]:
342
- """Read DOI list. Supports plain DOIs or TSV with DOI/PMID columns."""
525
+ """Read a worklist of DOIs.
526
+
527
+ Supports: plain DOI-per-line; TSV/CSV with a DOI header (optional PMID,
528
+ Title columns); and a Markdown pipe table with a DOI column. Each record is
529
+ {"doi", "pmid", "title"}.
530
+ """
531
+ text = Path(path).read_text(encoding="utf-8")
532
+ lines = text.splitlines()
533
+ first = next((ln for ln in lines
534
+ if ln.strip() and not ln.strip().startswith("#")), "")
535
+ low = first.lower()
536
+
537
+ # Markdown pipe table with a DOI column
538
+ if first.strip().startswith("|") and "doi" in low:
539
+ return _records_from_markdown(lines)
540
+
541
+ # Delimited (TSV or CSV) with a DOI header
542
+ if "doi" in low and ("\t" in first or "," in first):
543
+ delimiter = "\t" if "\t" in first else ","
544
+ body = "\n".join(ln for ln in lines if not ln.strip().startswith("#"))
545
+ reader = csv.DictReader(io.StringIO(body), delimiter=delimiter)
546
+ return _records_from_dictrows(reader)
547
+
548
+ # Plain text: one DOI per line
343
549
  records = []
344
- with open(path, encoding="utf-8") as f:
345
- first_line = f.readline().strip()
346
- f.seek(0)
347
-
348
- # TSV with header containing DOI column
349
- if "\t" in first_line and "doi" in first_line.lower():
350
- import csv
351
- reader = csv.DictReader(f, delimiter="\t")
352
- for row in reader:
353
- doi = ""
354
- pmid = ""
355
- for k, v in row.items():
356
- if k.lower().strip() == "doi":
357
- doi = (v or "").strip()
358
- elif k.lower().strip() == "pmid":
359
- pmid = (v or "").strip()
360
- if doi:
361
- records.append({"doi": doi, "pmid": pmid})
362
- else:
363
- # Plain text: one DOI per line
364
- for line in f:
365
- line = line.strip()
366
- if line and not line.startswith("#"):
367
- records.append({"doi": line, "pmid": ""})
550
+ for line in lines:
551
+ line = line.strip()
552
+ if line and not line.startswith("#"):
553
+ records.append({"doi": line, "pmid": "", "title": ""})
368
554
  return records
369
555
 
370
556
 
@@ -372,11 +558,15 @@ def main():
372
558
  parser = argparse.ArgumentParser(
373
559
  description="Batch download open-access PDFs by DOI.")
374
560
  parser.add_argument("input", type=Path,
375
- help="File with DOIs (one per line, or TSV with DOI column)")
561
+ help="Worklist: DOIs (one per line) or TSV/CSV/Markdown "
562
+ "with a DOI column (optional PMID, Title)")
376
563
  parser.add_argument("-o", "--output", type=Path, default=Path("pdfs"),
377
564
  help="Output directory (default: pdfs/)")
378
565
  parser.add_argument("-e", "--email", required=True,
379
566
  help="Contact email (required by Unpaywall TOS)")
567
+ parser.add_argument("--report", type=Path, default=None,
568
+ help="Path for the JSON retrieval report "
569
+ "(default: <output>/retrieval_report.json)")
380
570
  parser.add_argument("-v", "--verbose", action="store_true",
381
571
  help="Show debug messages")
382
572
  args = parser.parse_args()
@@ -387,33 +577,62 @@ def main():
387
577
  )
388
578
 
389
579
  args.output.mkdir(parents=True, exist_ok=True)
580
+ report_path = args.report or (args.output / "retrieval_report.json")
390
581
  records = read_doi_file(args.input)
391
582
  print(f"Loaded {len(records)} DOIs from {args.input}")
392
583
 
393
- stats = {"oa": 0, "pmc": 0, "fail": 0, "skip": 0}
584
+ stats = {"arxiv": 0, "oa": 0, "pmc": 0, "fail": 0, "skip": 0}
585
+ results: dict[str, tuple[str, str]] = {}
394
586
 
395
587
  for i, rec in enumerate(records, 1):
396
588
  doi = rec["doi"]
397
589
  pmid = rec.get("pmid", "")
398
590
  print(f" [{i}/{len(records)}] {doi}", end=" … ", flush=True)
399
591
 
400
- status = process_doi(doi, args.output, args.email, pmid)
592
+ status, source = process_doi(doi, args.output, args.email, pmid)
593
+ results[doi] = (status, source)
401
594
  stats[status] += 1
402
595
 
403
- labels = {"oa": "OK (OA)", "pmc": "OK (PMC)",
596
+ labels = {"arxiv": "OK (arXiv)", "oa": "OK (OA)", "pmc": "OK (PMC)",
404
597
  "fail": "FAIL", "skip": "SKIP"}
405
598
  print(labels[status])
406
599
  time.sleep(0.5)
407
600
 
601
+ # Best-effort title cross-check on successful downloads (needs pdftotext).
602
+ extracted: dict[str, str] = {}
603
+ have_titles = any(r.get("title") for r in records)
604
+ if have_titles and shutil.which("pdftotext"):
605
+ for rec in records:
606
+ doi = rec["doi"]
607
+ if not rec.get("title"):
608
+ continue
609
+ status, _ = results.get(doi, ("fail", ""))
610
+ if status not in RETRIEVED_STATUSES:
611
+ continue
612
+ path = args.output / f"{safe_doi_name(doi)}.pdf"
613
+ if path.exists():
614
+ text = extract_pdf_text(path)
615
+ if text:
616
+ extracted[doi] = text
617
+
618
+ report = build_report(records, results, args.output, extracted)
619
+ report_path.parent.mkdir(parents=True, exist_ok=True)
620
+ report_path.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
621
+
408
622
  print(f"\n--- Summary ---")
623
+ print(f" arXiv: {stats['arxiv']}")
409
624
  print(f" OA: {stats['oa']}")
410
625
  print(f" PMC: {stats['pmc']}")
411
626
  print(f" Failed: {stats['fail']}")
412
627
  print(f" Skipped: {stats['skip']}")
413
- total = stats["oa"] + stats["pmc"] + stats["fail"]
628
+ total = stats["arxiv"] + stats["oa"] + stats["pmc"] + stats["fail"]
414
629
  if total > 0:
415
- pct = (stats["oa"] + stats["pmc"]) / total * 100
630
+ pct = (stats["arxiv"] + stats["oa"] + stats["pmc"]) / total * 100
416
631
  print(f" Success: {pct:.0f}%")
632
+ mismatches = report["counts"]["title_mismatch"]
633
+ if mismatches:
634
+ print(f" Title mismatches flagged: {mismatches} (see report)")
635
+ print(f" Report: {report_path}")
417
636
 
418
637
  # Write failed DOIs for manual retrieval
419
638
  if stats["fail"] > 0:
@@ -422,10 +641,10 @@ def main():
422
641
  f.write("# DOIs needing manual retrieval\n")
423
642
  f.write("# Options: institutional access, ILL\n\n")
424
643
  for rec in records:
425
- safe = re.sub(r"[^\w\-.]", "_", rec["doi"])
426
- pdf = args.output / f"{safe}.pdf"
644
+ doi = rec["doi"]
645
+ pdf = args.output / f"{safe_doi_name(doi)}.pdf"
427
646
  if not existing_pdf_ok(pdf):
428
- f.write(f"{rec['doi']}\n")
647
+ f.write(f"{doi}\n")
429
648
  print(f" Manual list: {fail_path}")
430
649
 
431
650
 
@@ -0,0 +1,14 @@
1
+ {
2
+ "counts": {
3
+ "total": 4,
4
+ "retrieved": 3,
5
+ "not_retrieved": 1,
6
+ "title_mismatch": 1
7
+ },
8
+ "items": [
9
+ {"doi": "10.1111/valid.match", "status": "oa", "source": "unpaywall", "title_match": "match"},
10
+ {"doi": "10.2222/label.mismatch", "status": "oa", "source": "openalex", "title_match": "mismatch"},
11
+ {"doi": "10.3333/no.text", "status": "pmc", "source": "pmc", "title_match": "unavailable"},
12
+ {"doi": "10.9999/not.retrieved", "status": "fail", "source": "", "title_match": "unavailable"}
13
+ ]
14
+ }
@@ -0,0 +1,4 @@
1
+ {
2
+ "10.1111/valid.match": "Deep Learning for Pulmonary Nodule Detection on Chest CT\nAbstract: we present a convolutional neural network trained to detect pulmonary nodules.",
3
+ "10.2222/label.mismatch": "Genome-wide association study of type 2 diabetes in an Asian cohort\nIntroduction: we genotyped participants to identify susceptibility loci."
4
+ }
@@ -0,0 +1,6 @@
1
+ {
2
+ "10.1111/valid.match": ["oa", "unpaywall"],
3
+ "10.2222/label.mismatch": ["oa", "openalex"],
4
+ "10.3333/no.text": ["pmc", "pmc"],
5
+ "10.9999/not.retrieved": ["fail", ""]
6
+ }
@@ -0,0 +1,86 @@
1
+ #!/usr/bin/env python3
2
+ """Offline, network-free challenge for fetch_oa.build_report + title tri-state.
3
+
4
+ Loads committed fixtures (worklist.tsv, results.json, extracted_text.json),
5
+ fabricates stub PDFs in a temp dir for the "retrieved" DOIs, runs
6
+ fetch_oa.build_report, and asserts the resulting projection equals
7
+ expected/projection.json. Exercises read_doi_file (TSV+title), build_report,
8
+ and classify_title_match (match / mismatch / unavailable) without touching the
9
+ network or pdftotext. Stdlib-only.
10
+ """
11
+ import importlib.util
12
+ import json
13
+ import sys
14
+ import tempfile
15
+ from pathlib import Path
16
+
17
+ HERE = Path(__file__).resolve().parent
18
+ ENGINE = HERE.parent / "fetch_oa.py"
19
+
20
+
21
+ def load_engine():
22
+ spec = importlib.util.spec_from_file_location("fetch_oa", ENGINE)
23
+ mod = importlib.util.module_from_spec(spec)
24
+ spec.loader.exec_module(mod)
25
+ return mod
26
+
27
+
28
+ def projection(report: dict) -> dict:
29
+ items = sorted(
30
+ ({"doi": i["doi"], "status": i["status"], "source": i["source"],
31
+ "title_match": i["title_match"]} for i in report["items"]),
32
+ key=lambda x: x["doi"],
33
+ )
34
+ return {"counts": report["counts"], "items": items}
35
+
36
+
37
+ def main() -> int:
38
+ assert ENGINE.exists(), f"ENV-ERR: {ENGINE} missing"
39
+ m = load_engine()
40
+
41
+ records = m.read_doi_file(HERE / "worklist.tsv")
42
+ results = {k: tuple(v) for k, v in
43
+ json.loads((HERE / "results.json").read_text()).items()}
44
+ extracted = json.loads((HERE / "extracted_text.json").read_text())
45
+ expected = json.loads((HERE / "expected" / "projection.json").read_text())
46
+
47
+ fails = []
48
+
49
+ def check(label, cond):
50
+ print(f" {'PASS' if cond else 'FAIL'} {label}")
51
+ if not cond:
52
+ fails.append(label)
53
+
54
+ with tempfile.TemporaryDirectory() as tmp:
55
+ outdir = Path(tmp)
56
+ for rec in records:
57
+ status, _ = results.get(rec["doi"], ("fail", ""))
58
+ if status in m.RETRIEVED_STATUSES:
59
+ stub = outdir / f"{m.safe_doi_name(rec['doi'])}.pdf"
60
+ stub.write_bytes(b"%PDF-1.4\n" + b"0" * (11 * 1024))
61
+ report = m.build_report(records, results, outdir, extracted)
62
+ proj = projection(report)
63
+
64
+ by_doi = {i["doi"]: i for i in proj["items"]}
65
+ check("valid -> title_match 'match'",
66
+ by_doi["10.1111/valid.match"]["title_match"] == "match")
67
+ check("mislabel -> title_match 'mismatch'",
68
+ by_doi["10.2222/label.mismatch"]["title_match"] == "mismatch")
69
+ check("no-text -> title_match 'unavailable'",
70
+ by_doi["10.3333/no.text"]["title_match"] == "unavailable")
71
+ check("missing -> status 'fail'",
72
+ by_doi["10.9999/not.retrieved"]["status"] == "fail")
73
+ check("counts.retrieved == 3", proj["counts"]["retrieved"] == 3)
74
+ check("counts.not_retrieved == 1", proj["counts"]["not_retrieved"] == 1)
75
+ check("counts.title_mismatch == 1", proj["counts"]["title_mismatch"] == 1)
76
+ check("projection matches expected/projection.json", proj == expected)
77
+
78
+ if fails:
79
+ print(f"FAILURES: {len(fails)}")
80
+ return 1
81
+ print("ALL PASS")
82
+ return 0
83
+
84
+
85
+ if __name__ == "__main__":
86
+ sys.exit(main())
@@ -0,0 +1,6 @@
1
+ #!/usr/bin/env bash
2
+ # Network-free challenge: fetch_oa report builder + title-match tri-state.
3
+ # Mirrors CI usage: `bash skills/fulltext-retrieval/fetch_oa_report_challenge/verify.sh`
4
+ set -euo pipefail
5
+ DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
6
+ python3 "$DIR/run_challenge.py"
@@ -0,0 +1,5 @@
1
+ DOI PMID Title
2
+ 10.1111/valid.match 1001 Deep learning for pulmonary nodule detection on chest CT
3
+ 10.2222/label.mismatch 1002 Automated segmentation of cardiac MRI using transformers
4
+ 10.3333/no.text 1003 Federated learning for multi-center radiology models
5
+ 10.9999/not.retrieved 1004 A paywalled study with no open-access copy
@@ -0,0 +1,82 @@
1
+ /*
2
+ * Find Available PDF — batch trigger for Zotero (user-run snippet)
3
+ * ----------------------------------------------------------------
4
+ * Attaches full-text PDFs to library items using Zotero's OWN
5
+ * "Find Available PDF" resolver. This reuses whatever OpenURL resolver,
6
+ * institutional proxy, or library configuration YOU have already set in
7
+ * Zotero — so it typically retrieves far more than open-access-only resolvers,
8
+ * yet no credentials, proxy hosts, or institutional identifiers ever leave
9
+ * your Zotero client. Nothing here is hard-coded to any institution.
10
+ *
11
+ * HOW TO RUN
12
+ * 1. In Zotero, select the items (or open the collection) you want PDFs for.
13
+ * 2. Tools → Developer → Run JavaScript.
14
+ * 3. Paste this whole snippet and click Run.
15
+ * 4. The result panel prints a JSON summary: considered / attached /
16
+ * alreadyHadPDF / missing (DOIs still without a PDF).
17
+ *
18
+ * NO-CODE FALLBACK
19
+ * Select items → right-click → "Find Available PDF" does the same thing
20
+ * interactively. Use it if you prefer not to run a script.
21
+ *
22
+ * VERSION NOTE
23
+ * Zotero 7 exposes a batch Zotero.Attachments.addAvailablePDFs(items);
24
+ * Zotero 6 only has the per-item Zotero.Attachments.addAvailablePDF(item).
25
+ * This snippet prefers the batch call and falls back to per-item.
26
+ *
27
+ * NOTE: results are user-initiated and depend on your live Zotero session;
28
+ * they are NOT reproducible CI evidence. Record retrieved/not-retrieved
29
+ * outcomes from the printed summary into your retrieval report manually.
30
+ */
31
+
32
+ var pane = Zotero.getActiveZoteroPane();
33
+ var items = pane.getSelectedItems().filter(function (it) { return it.isRegularItem(); });
34
+
35
+ // Fall back to the whole selected collection if nothing is selected.
36
+ if (!items.length) {
37
+ var collection = pane.getSelectedCollection();
38
+ if (collection) {
39
+ items = collection.getChildItems().filter(function (it) { return it.isRegularItem(); });
40
+ }
41
+ }
42
+
43
+ function hasPDF(item) {
44
+ return item.getAttachments().some(function (id) {
45
+ var att = Zotero.Items.get(id);
46
+ if (!att) return false;
47
+ if (typeof att.isPDFAttachment === "function") return att.isPDFAttachment();
48
+ return att.attachmentContentType === "application/pdf";
49
+ });
50
+ }
51
+
52
+ var todo = items.filter(function (it) { return !hasPDF(it); });
53
+ var alreadyHadPDF = items.length - todo.length;
54
+
55
+ if (typeof Zotero.Attachments.addAvailablePDFs === "function") {
56
+ await Zotero.Attachments.addAvailablePDFs(todo); // Zotero 7 batch
57
+ } else {
58
+ for (let it of todo) { // Zotero 6 per-item
59
+ try {
60
+ await Zotero.Attachments.addAvailablePDF(it);
61
+ } catch (e) {
62
+ Zotero.debug("addAvailablePDF failed for item " + it.id + ": " + e);
63
+ }
64
+ }
65
+ }
66
+
67
+ var attached = 0;
68
+ var missing = [];
69
+ for (let it of todo) {
70
+ if (hasPDF(it)) {
71
+ attached++;
72
+ } else {
73
+ missing.push(it.getField("DOI") || it.getField("title") || ("itemID:" + it.id));
74
+ }
75
+ }
76
+
77
+ return JSON.stringify({
78
+ considered: items.length,
79
+ attached: attached,
80
+ alreadyHadPDF: alreadyHadPDF,
81
+ missing: missing
82
+ }, null, 2);
@@ -8,12 +8,14 @@ when_to_use: "Batch-download open-access full-text PDFs from a DOI list; optiona
8
8
  when_NOT_to_use: "Finding or verifying citations (use search-lit / verify-refs). Retrieving paywalled or non-open-access content."
9
9
 
10
10
  inputs:
11
- - path: "DOI list (.txt, one per line, or .tsv with a DOI column)"
11
+ - path: "DOI worklist (.txt one-per-line, or .tsv/.csv/.md with a DOI column; optional PMID, Title)"
12
12
  schema: csv
13
13
  required: true
14
14
  outputs:
15
15
  - path: "downloaded open-access PDFs (pdfs/)"
16
+ - path: "pdfs/retrieval_report.json (per-DOI status/source/title_match tri-state)"
16
17
  - path: "optional PDF-to-Markdown conversions"
18
+ - path: "references/find_available_pdf.js (user-run Zotero 'Find Available PDF' batch snippet)"
17
19
 
18
20
  deterministic_scripts:
19
21
  - fetch_oa.py
@@ -35,8 +37,11 @@ safety_boundaries:
35
37
  - "Validates each download (>=10 KB and a %PDF- header) before accepting it."
36
38
  known_limitations:
37
39
  - "Only open-access content is retrievable; non-OA DOIs fail by design rather than fetching from unauthorized sources."
40
+ - "Higher-yield in-library retrieval (find_available_pdf.js) is user-initiated inside Zotero and uses the user's own proxy/OpenURL config; it is not reproducible CI evidence."
41
+ - "Title cross-check is best-effort: it needs a Title column plus pdftotext (poppler); otherwise title_match is 'unavailable'. A mismatch is flagged, never auto-rejected."
38
42
  - "PDF-to-Markdown conversion requires the optional pymupdf4llm dependency (AGPL-3.0 or commercial license)."
39
43
  validation_commands:
40
- - "python fetch_oa.py dois.txt -o pdfs/ -e <email> --verbose # per-DOI source trace"
44
+ - "bash fetch_oa_report_challenge/verify.sh # offline report-builder + title tri-state (CI-wired)"
45
+ - "python fetch_oa.py dois.txt -o pdfs/ -e <email> --report pdfs/retrieval_report.json --verbose # per-DOI source trace"
41
46
  - "verify each output begins with %PDF- and is at least 10 KB"
42
47
  evidence_surface: bundled_script
@@ -65,6 +65,9 @@ Direct hand edits to `refs.bib` are drift — revert on sight.
65
65
  ▼ Phase 2.5: refs.bib snapshot refresh
66
66
  Trigger Better BibTeX auto-export → verify manuscript/_src/refs.bib mtime updated
67
67
 
68
+ ▼ Phase 2.7: Fulltext Retrieval (opt-in)
69
+ Disk OA PDFs via /fulltext-retrieval + in-library via find_available_pdf.js → reconcile report
70
+
68
71
  ▼ Phase 3: Obsidian Literature Notes
69
72
  Create Literature/{citekey}.md (empty note OK — fill later with highlights)
70
73
 
@@ -116,8 +119,16 @@ collection key for future use.
116
119
  For each entry:
117
120
 
118
121
  1. Use `zotero_search_items` to search by DOI or title — if already present, skip.
122
+ This search-first step is what prevents duplicates; `zotero_add_by_doi` does **not**
123
+ dedupe by itself (it fetches CrossRef and creates the item), so never skip the search.
119
124
  2. Otherwise call `zotero_add_by_doi` (when a DOI is available) or
120
125
  `zotero_add_by_url` (falling back to the PubMed URL when no DOI is available).
126
+ - `zotero_add_by_doi` accepts an `attach_mode` argument that governs the **OA child-PDF
127
+ attach attempt at add time** (the installed server treats `linked_url` as "bookmark the
128
+ PDF URL"; other values download/import). Set it when you want a PDF attached during the
129
+ add. Exact accepted values are server-version-specific — verify against the connected
130
+ server. Do **not** use `zotero_add_from_file` to attach a PDF to an item added here: it
131
+ has no parent-item argument and would create a duplicate parent item.
121
132
  3. Use `zotero_manage_collections` to place the item in the project collection.
122
133
 
123
134
  ### Step 2.3: Result report
@@ -214,6 +225,62 @@ If refresh failed, set `refs_bib_refreshed: false` and include `reason`. `/verif
214
225
 
215
226
  ---
216
227
 
228
+ ## Phase 2.7: Fulltext Retrieval (opt-in, owner-only)
229
+
230
+ **Run only when the user asks for full text** (e.g. "download the PDFs", "fetch full
231
+ text", or a worklist supplied with that intent). Default `/lit-sync` stays metadata-only
232
+ and network-light — do not auto-run this phase. Runs after items are in Zotero (Phase 2)
233
+ and the snapshot is verified (Phase 2.5), before Obsidian notes (Phase 3).
234
+
235
+ There are two complementary retrieval routes; offer both and reconcile them in one report:
236
+
237
+ ### Route A — disk OA PDFs (for downstream skills)
238
+
239
+ Delegate to the `/fulltext-retrieval` engine (do **not** re-implement the OA cascade or
240
+ import its code; invoke it by path). Resolve the engine as:
241
+
242
+ ```bash
243
+ ENGINE="${MEDSCI_SKILLS_ROOT:-$HOME/workspace/medsci-skills}/skills/fulltext-retrieval/fetch_oa.py"
244
+ python3 "$ENGINE" <worklist> -o pdfs/ -e <contact-email> --report pdfs/retrieval_report.json
245
+ ```
246
+
247
+ `<worklist>` is the DOI/PMID(/Title) list — the Phase-1 `.bib` DOIs, the worklist supplied
248
+ in the standalone mode below, or the project collection's DOIs. Output: `pdfs/*.pdf` for
249
+ `/meta-analysis`, `/obsidian-paper-vault`, and `pdf_to_md.py`, plus
250
+ `pdfs/retrieval_report.json` (per-DOI `status`/`source`/`title_match`).
251
+
252
+ ### Route B — in-library PDFs (Zotero-native, higher yield, proxy-aware)
253
+
254
+ Emit `${MEDSCI_SKILLS_ROOT:-$HOME/workspace/medsci-skills}/skills/fulltext-retrieval/references/find_available_pdf.js`
255
+ for the user to paste into Zotero (*Tools → Developer → Run JavaScript*) with the project
256
+ collection selected. It triggers Zotero's own `addAvailablePDF`/`addAvailablePDFs`, which
257
+ reuse the **user's** OpenURL resolver / institutional proxy — so it typically retrieves more
258
+ than OA-only, while **no credentials or institutional identifiers enter this skill**. The
259
+ no-code equivalent is right-click → "Find Available PDF". This route is user-initiated and
260
+ session-dependent; record its `{attached, missing}` summary from the printed JSON.
261
+
262
+ ### Report
263
+
264
+ Merge Route A's `pdfs/retrieval_report.json` (and the user-reported Route B summary) into
265
+ `references/fulltext_retrieval.json` (owner of this file is `/lit-sync`):
266
+
267
+ ```json
268
+ {
269
+ "schema_version": 1,
270
+ "retrieved_oa_disk": [{"doi": "...", "source": "unpaywall", "file": "...", "title_match": "match"}],
271
+ "retrieved_zotero_native": [{"doi": "...", "via": "addAvailablePDF"}],
272
+ "not_retrieved": [{"doi": "...", "journal": "..."}],
273
+ "institutional_fallback": ["<DOIs needing institutional access / ILL / author contact>"],
274
+ "title_mismatch_flagged": ["<DOIs whose downloaded PDF title did not match>"]
275
+ }
276
+ ```
277
+
278
+ Also append a short `fulltext` block (counts) to `references/zotero_collection.json`.
279
+ `not_retrieved` DOIs are candidates for institutional access, interlibrary loan, or author
280
+ contact — never bypass paywalls or access controls from this skill.
281
+
282
+ ---
283
+
217
284
  ## Phase 3: Obsidian Literature Notes
218
285
 
219
286
  ### Step 3.1: Check existing literature notes
@@ -414,16 +481,23 @@ curl -s "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi?db=pubmed&i
414
481
  | jq -r '.result | to_entries[] | select(.key != "uids") | "\(.value.uid)\t\(.value.elocationid)\t\(.value.title)"'
415
482
  ```
416
483
 
417
- For each resolved DOI call `zotero_add_by_doi` (auto-dedup by DOI). For items
418
- already in the library (returned as "skipped" or detected via `zotero_search_items`
419
- by DOI), use `zotero_manage_collections` to attach them to the project collection
420
- **without re-adding** re-adding by URL/PubMed-URL would bypass DOI dedup and
421
- create duplicates. Record both `added` and `existing` items in
422
- `references/zotero_collection.json`.
484
+ For each resolved DOI, search-first with `zotero_search_items`, then call
485
+ `zotero_add_by_doi` the search is what dedupes (add-by-doi alone does not). For items
486
+ already in the library (detected via `zotero_search_items` by DOI), use
487
+ `zotero_manage_collections` to attach them to the project collection **without re-adding**
488
+ re-adding by URL/PubMed-URL would bypass the search dedup and create duplicates. Record both
489
+ `added` and `existing` items in `references/zotero_collection.json`.
423
490
 
424
491
  If a PMID has no DOI in PubMed (rare; older papers, non-indexed), fall back to
425
492
  `zotero_add_by_url` with the PubMed URL and mark the entry as `no_doi: true`.
426
493
 
494
+ ### Worklist ingestion (DOI/PMID/Title; no .bib)
495
+ When the user supplies a worklist file (a `.tsv`/`.csv`/`.md` table with a `DOI` column,
496
+ optional `PMID`/`Title`, or a plain DOI-per-line list — e.g. an SR include set), enter
497
+ Phase 2 directly from it: resolve any PMID-only rows to DOIs (esummary above), then run the
498
+ search-first dedupe + add loop. The same worklist file feeds Phase 2.7 Route A
499
+ (`fetch_oa.py` reads `.tsv`/`.csv`/`.md`/plain natively), so no reformatting is needed.
500
+
427
501
  ---
428
502
 
429
503
  ## Safety Rules
@@ -439,6 +513,12 @@ If a PMID has no DOI in PubMed (rare; older papers, non-indexed), fall back to
439
513
  collection is created.
440
514
  6. **Never write `refs.bib` directly.** Only Better BibTeX auto-export may write that file. If auto-export is broken, fix the Zotero setup rather than writing the file from this skill.
441
515
  7. **Owner-only execution.** If the current user is a collaborator (no Zotero access per `SSOT.yaml` `reference_manager.required_for`), abort with instructions to flag `[@NEW:topic]` placeholders in the manuscript and notify the owner.
516
+ 8. **Fulltext boundary (Phase 2.7).** Retrieve full text only via OA APIs (the
517
+ `/fulltext-retrieval` engine) and the user-run Zotero "Find Available PDF" snippet
518
+ (which uses the user's own proxy config). Never automate authenticated browser sessions,
519
+ never bypass paywalls/access controls, and never hard-code institutional proxies,
520
+ credentials, or hosts into this skill. `not_retrieved` items are routed to institutional
521
+ access / ILL / author contact, not worked around.
442
522
 
443
523
  ## Anti-Hallucination
444
524
 
@@ -8,6 +8,7 @@ when_to_use:
8
8
  - After /search-lit when verified candidates need to flow into Zotero + refs.bib
9
9
  - Cross-cutting concept-note extraction from accumulated literature into Obsidian
10
10
  - Refreshing manuscript/_src/refs.bib via Better BibTeX auto-export before a build
11
+ - Opt-in full-text retrieval (Phase 2.7) — disk OA PDFs via /fulltext-retrieval + in-library "Find Available PDF"
11
12
  when_NOT_to_use:
12
13
  - Pure literature search without sync (use /search-lit)
13
14
  - Hand-editing refs.bib (BBT is the sole writer; never touch the file directly)
@@ -17,10 +18,11 @@ inputs:
17
18
  - PMID list, DOI list, or Zotero collection name
18
19
  outputs:
19
20
  - references/zotero_collection.json
21
+ - references/fulltext_retrieval.json # SOLE WRITER; opt-in Phase 2.7 retrieval report
20
22
  - manuscript/_src/refs.bib # SOLE WRITER via Better BibTeX auto-export "Keep updated"
21
23
  - obsidian_literature_notes
22
24
  deterministic_scripts:
23
- - none_required # leverages Zotero MCP + Better BibTeX auto-export GUI
25
+ - none_required # leverages Zotero MCP + Better BibTeX auto-export GUI; Phase 2.7 invokes /fulltext-retrieval fetch_oa.py
24
26
  side_effects:
25
27
  - may_update_zotero
26
28
  - may_write_obsidian_notes
@@ -263,9 +263,9 @@ Total: 13 references
263
263
 
264
264
  If a Zotero MCP server is available, integrate search results with the user's library:
265
265
 
266
- 1. **Add papers to Zotero**: Use `zotero_add_by_doi` for DOI-based import (auto-downloads OA PDFs).
267
- 2. **Organize into collections**: Use `zotero_manage_collections` to file into the relevant project collection.
268
- 3. **Check for duplicates**: Use `zotero_search_items` to avoid adding papers already in the library.
266
+ 1. **Check for duplicates first**: Use `zotero_search_items` (by DOI) to skip papers already in the library — this search-first step is what dedupes; `zotero_add_by_doi` does not dedupe on its own.
267
+ 2. **Add papers to Zotero**: Use `zotero_add_by_doi` for DOI-based import (its `attach_mode` argument governs the OA PDF attach attempt at add time).
268
+ 3. **Organize into collections**: Use `zotero_manage_collections` to file into the relevant project collection.
269
269
  4. **Leverage annotations**: Use `zotero_get_annotations` to reference the user's prior reading notes.
270
270
  5. **Write sync audit**: Record collection key, added/skipped/failed counts, and
271
271
  unsynced entries in `references/zotero_collection.json` so Zotero status is
@@ -277,77 +277,32 @@ If a Zotero MCP server is available, integrate search results with the user's li
277
277
 
278
278
  ### Phase 5: Full-Text Retrieval
279
279
 
280
- After identifying relevant papers, retrieve full-text PDFs for detailed review.
281
- This is especially important for meta-analyses where data extraction requires full text.
280
+ Full-text PDF retrieval is **delegated to `/fulltext-retrieval`** the single authored
281
+ home of the open-access cascade (arXiv Unpaywall PMC OpenAlex → Crossref → landing
282
+ page, each validated with a `%PDF-` header + ≥10 KB size). Do **not** re-implement OA
283
+ fetching here.
282
284
 
283
- #### Phase 5a: Open Access Auto-Retrieval
285
+ Pass the verified candidate DOIs from `references/library.bib`:
284
286
 
285
- Try sources in order of reliability:
286
-
287
- 1. **Unpaywall API** (highest quality OA links):
288
- ```python
289
- import os, requests
290
- email = os.environ.get("UNPAYWALL_EMAIL", "user@example.com")
291
- url = f"https://api.unpaywall.org/v2/{doi}?email={email}"
292
- r = requests.get(url).json()
293
- if r.get("best_oa_location", {}).get("url_for_pdf"):
294
- pdf_url = r["best_oa_location"]["url_for_pdf"]
295
- ```
296
-
297
- 2. **PubMed Central (PMC)**:
298
- - Convert PMID to PMCID via NCBI ID Converter
299
- - Download from PMC OA service: `https://www.ncbi.nlm.nih.gov/pmc/articles/PMC{id}/pdf/`
300
-
301
- 3. **OpenAlex API** (additional OA discovery):
302
- ```python
303
- url = f"https://api.openalex.org/works/https://doi.org/{doi}"
304
- # Requires polite pool: add email in User-Agent header or mailto= param
305
- r = requests.get(url, headers={"User-Agent": f"MyApp/1.0 (mailto:{email})"}).json()
306
- oa_url = r.get("open_access", {}).get("oa_url")
307
- ```
308
-
309
- 4. **CrossRef landing page**: Follow `https://api.crossref.org/works/{doi}` → publisher link
310
- → scrape `<meta name="citation_pdf_url">` tag
311
-
312
- #### Phase 5b: Alternative Sources
313
-
314
- Some researchers use alternative access methods for paywalled content.
315
- **Users are responsible for ensuring compliance with their institutional access policies.**
316
-
317
- If an environment variable (e.g., `SCIHUB_BASE`) is set, the skill may use it as an
318
- alternative PDF source. No specific URLs are provided here — users configure this themselves.
319
-
320
- Other options:
321
- - **Institutional proxy/VPN**: Access publisher sites through institutional EZproxy or VPN
322
- - **Interlibrary loan (ILL)**: Request through library services for papers not otherwise available
323
- - **Author contact**: Email corresponding authors for preprints
324
-
325
- #### PDF Validation
287
+ ```bash
288
+ ENGINE="${MEDSCI_SKILLS_ROOT:-$HOME/workspace/medsci-skills}/skills/fulltext-retrieval/fetch_oa.py"
289
+ # extract DOIs from references/library.bib dois.txt (one per line)
290
+ python3 "$ENGINE" dois.txt -o pdfs/ -e <contact-email> --report pdfs/retrieval_report.json
291
+ ```
326
292
 
327
- Always validate downloaded files before use:
293
+ For Zotero-resident PDFs and higher-yield, proxy-aware retrieval, use `/lit-sync` Phase 2.7,
294
+ which also invokes `/fulltext-retrieval` and triggers Zotero's native "Find Available PDF".
328
295
 
329
- ```python
330
- def is_valid_pdf(filepath):
331
- """Check that a downloaded file is actually a PDF, not an HTML redirect."""
332
- import os
333
- if os.path.getsize(filepath) < 10240: # < 10KB is likely a stub/redirect
334
- return False
335
- with open(filepath, 'rb') as f:
336
- header = f.read(5)
337
- return header == b'%PDF-'
338
- ```
296
+ #### Alternative sources (legitimate only)
339
297
 
340
- Additional checks:
341
- - Verify HTTP `Content-Type: application/pdf` header before saving
342
- - Files under 10KB are almost always HTML login/redirect pages, not real PDFs
343
- - Some publishers return CAPTCHA pages — these fail the `%PDF-` check
298
+ For DOIs that open access cannot reach (listed in `pdfs/manual_needed.txt`):
344
299
 
345
- #### Rate Limiting
300
+ - **Institutional access / proxy / VPN** — through your library's own subscriptions.
301
+ - **Interlibrary loan (ILL)** — request via library services.
302
+ - **Author contact** — email the corresponding author for a copy or preprint.
346
303
 
347
- - Unpaywall: Polite pool (no hard limit with email parameter)
348
- - OpenAlex: Include email in User-Agent for polite pool access
349
- - NCBI/PMC: 3 requests/sec without API key, 10/sec with `NCBI_API_KEY`
350
- - General: 2-second minimum interval between requests to any single host
304
+ Never bypass paywalls or publisher access controls, and do not configure unauthorized
305
+ PDF mirrors. Rate limits and PDF validation are handled inside `/fulltext-retrieval`.
351
306
 
352
307
  ### Phase 6: Gap Analysis
353
308