pyPaperFlow 0.6.0__tar.gz → 0.6.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/PKG-INFO +27 -1
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/README.md +26 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/README_zh.md +26 -0
- pypaperflow-0.6.1/src/pyPaperFlow/__init__.py +1 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/biorxiv_fetcher.py +134 -3
- pypaperflow-0.6.0/src/pyPaperFlow/__init__.py +0 -1
- pypaperflow-0.6.0/tests/test_arxiv_fulltext.py +0 -47
- pypaperflow-0.6.0/tests/test_biorxiv_fulltext.py +0 -101
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/.claude/settings.local.json +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/.github/workflows/docs.yml +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/.gitignore +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/LICENSE +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/mkdoc_site/index.md +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/mkdocs.yml +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/pyproject.toml +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/requirements-docs.txt +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/scripts/sync_docs.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/cli.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/cloak_fallback.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/cloak_pdf.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/github_export.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/mineru_parser.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/pdf_fetch.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/undetected_fallback.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/undetected_pdf.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/arxiv_fetcher.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/chemrxiv_fetcher.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/europepmc_fetcher.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/source_merge.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/source_models.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/source_utils.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/pubmed/__init__.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/pubmed/pubmed_fetcher.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/pubmed/pubmed_merger.py +0 -0
- {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pyPaperFlow
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.1
|
|
4
4
|
Summary: Automated paper fetching and analysis platform.
|
|
5
5
|
Project-URL: Homepage, https://github.com/MaybeBio/pyPaperFlow
|
|
6
6
|
Project-URL: Issues, https://github.com/MaybeBio/pyPaperFlow/issues
|
|
@@ -112,6 +112,7 @@ This tool is designed to `complement rather than replace` reference management s
|
|
|
112
112
|
|
|
113
113
|
- **Automated Retrieval from Multiple Sources**: Automatically search and retrieve paper metadata and full-text records from `PubMed/Medline, arXiv, medRxiv, chemRxiv and bioRxiv`. The repository focuses primarily on biomedical research and computational interdisciplinary fields (`Biomedicine + Computational Biology`).
|
|
114
114
|
- **Full-Text Access**: Enable automatic downloading of open-access full texts in XML/Text format from `PMC`. For preprints and other publications without accessible PMC full texts, alternative acquisition modules are integrated to fetch `original PDFs`, with `Sci-Hub` set as the fallback provider.
|
|
115
|
+
- **Preprint Full-Text Fetch (no PDF parsing)**: For preprints without an OA PDF, dedicated methods return clean section-headed text directly — `ArxivFetcher.fetch_full_text(arxiv_id)` reads ar5iv-rendered HTML (arXiv LaTeX → HTML), and `BioRxivFetcher.fetch_full_text(doi)` / `EuropePMCFullText.full_text_xml(doi)` read JATS full-text XML from Europe PMC (bioRxiv / medRxiv and other DOI-indexed preprints).
|
|
115
116
|
- **Structured Storage**:
|
|
116
117
|
- **Metadata**: Preserved in well-structured detailed JSON files.
|
|
117
118
|
- **Full Text**: Stored in multiple formats including parsed JSON and Markdown for versatile downstream usage — JSON for programmatic data analysis, and Markdown optimized for LLM comprehension and processing.
|
|
@@ -1450,6 +1451,31 @@ In theory, all DOI‑driven literature workflows can be standardised following t
|
|
|
1450
1451
|
|
|
1451
1452
|
> Modules dedicated to the aforementioned preprint platforms are still under development and refinement. Preprint‑related subcommands are provided for testing purposes only. For detailed test cases, refer to [Cases](./docs/Cases.md)
|
|
1452
1453
|
|
|
1454
|
+
#### Preprint full-text fetch (Python API)
|
|
1455
|
+
|
|
1456
|
+
For preprints without an open-access PDF, each fetcher exposes a `fetch_full_text()` method that returns clean section-headed text without any PDF parsing:
|
|
1457
|
+
|
|
1458
|
+
```python
|
|
1459
|
+
from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher
|
|
1460
|
+
from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher
|
|
1461
|
+
from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
|
|
1462
|
+
|
|
1463
|
+
# arXiv → ar5iv rendered HTML (LaTeX → HTML)
|
|
1464
|
+
arxiv = ArxivFetcher(root_dir="./papers")
|
|
1465
|
+
text = arxiv.fetch_full_text("1706.03762") # "" on failure
|
|
1466
|
+
|
|
1467
|
+
# bioRxiv / medRxiv → Europe PMC fullTextXML
|
|
1468
|
+
biorxiv = BioRxivFetcher(root_dir="./papers", platform="biorxiv")
|
|
1469
|
+
text = biorxiv.fetch_full_text("10.1101/2023.06.22.546069")
|
|
1470
|
+
|
|
1471
|
+
# Any DOI-indexed preprint → Europe PMC fullTextXML directly
|
|
1472
|
+
epmc = EuropePMCFullText()
|
|
1473
|
+
xml = epmc.full_text_xml("10.1101/2023.06.22.546069")
|
|
1474
|
+
epmc.close()
|
|
1475
|
+
```
|
|
1476
|
+
|
|
1477
|
+
All three return an empty string `""` on failure, so callers can gracefully fall back to the abstract. The returned text is section-headed (`## Section`) plain text ready for LLM input.
|
|
1478
|
+
|
|
1453
1479
|
### 6. Critical Reading and Knowledge Graph Analysis: Downstream End‑Use
|
|
1454
1480
|
|
|
1455
1481
|
Upon completing literature retrieval, parsing, and structured processing as outlined above, users obtain chapter‑organised Markdown files and structured JSON files, which serve as the fundamental inputs for subsequent critical reading and knowledge graph analysis.
|
|
@@ -83,6 +83,7 @@ This tool is designed to `complement rather than replace` reference management s
|
|
|
83
83
|
|
|
84
84
|
- **Automated Retrieval from Multiple Sources**: Automatically search and retrieve paper metadata and full-text records from `PubMed/Medline, arXiv, medRxiv, chemRxiv and bioRxiv`. The repository focuses primarily on biomedical research and computational interdisciplinary fields (`Biomedicine + Computational Biology`).
|
|
85
85
|
- **Full-Text Access**: Enable automatic downloading of open-access full texts in XML/Text format from `PMC`. For preprints and other publications without accessible PMC full texts, alternative acquisition modules are integrated to fetch `original PDFs`, with `Sci-Hub` set as the fallback provider.
|
|
86
|
+
- **Preprint Full-Text Fetch (no PDF parsing)**: For preprints without an OA PDF, dedicated methods return clean section-headed text directly — `ArxivFetcher.fetch_full_text(arxiv_id)` reads ar5iv-rendered HTML (arXiv LaTeX → HTML), and `BioRxivFetcher.fetch_full_text(doi)` / `EuropePMCFullText.full_text_xml(doi)` read JATS full-text XML from Europe PMC (bioRxiv / medRxiv and other DOI-indexed preprints).
|
|
86
87
|
- **Structured Storage**:
|
|
87
88
|
- **Metadata**: Preserved in well-structured detailed JSON files.
|
|
88
89
|
- **Full Text**: Stored in multiple formats including parsed JSON and Markdown for versatile downstream usage — JSON for programmatic data analysis, and Markdown optimized for LLM comprehension and processing.
|
|
@@ -1421,6 +1422,31 @@ In theory, all DOI‑driven literature workflows can be standardised following t
|
|
|
1421
1422
|
|
|
1422
1423
|
> Modules dedicated to the aforementioned preprint platforms are still under development and refinement. Preprint‑related subcommands are provided for testing purposes only. For detailed test cases, refer to [Cases](./docs/Cases.md)
|
|
1423
1424
|
|
|
1425
|
+
#### Preprint full-text fetch (Python API)
|
|
1426
|
+
|
|
1427
|
+
For preprints without an open-access PDF, each fetcher exposes a `fetch_full_text()` method that returns clean section-headed text without any PDF parsing:
|
|
1428
|
+
|
|
1429
|
+
```python
|
|
1430
|
+
from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher
|
|
1431
|
+
from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher
|
|
1432
|
+
from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
|
|
1433
|
+
|
|
1434
|
+
# arXiv → ar5iv rendered HTML (LaTeX → HTML)
|
|
1435
|
+
arxiv = ArxivFetcher(root_dir="./papers")
|
|
1436
|
+
text = arxiv.fetch_full_text("1706.03762") # "" on failure
|
|
1437
|
+
|
|
1438
|
+
# bioRxiv / medRxiv → Europe PMC fullTextXML
|
|
1439
|
+
biorxiv = BioRxivFetcher(root_dir="./papers", platform="biorxiv")
|
|
1440
|
+
text = biorxiv.fetch_full_text("10.1101/2023.06.22.546069")
|
|
1441
|
+
|
|
1442
|
+
# Any DOI-indexed preprint → Europe PMC fullTextXML directly
|
|
1443
|
+
epmc = EuropePMCFullText()
|
|
1444
|
+
xml = epmc.full_text_xml("10.1101/2023.06.22.546069")
|
|
1445
|
+
epmc.close()
|
|
1446
|
+
```
|
|
1447
|
+
|
|
1448
|
+
All three return an empty string `""` on failure, so callers can gracefully fall back to the abstract. The returned text is section-headed (`## Section`) plain text ready for LLM input.
|
|
1449
|
+
|
|
1424
1450
|
### 6. Critical Reading and Knowledge Graph Analysis: Downstream End‑Use
|
|
1425
1451
|
|
|
1426
1452
|
Upon completing literature retrieval, parsing, and structured processing as outlined above, users obtain chapter‑organised Markdown files and structured JSON files, which serve as the fundamental inputs for subsequent critical reading and knowledge graph analysis.
|
|
@@ -82,6 +82,7 @@
|
|
|
82
82
|
|
|
83
83
|
- **多来源自动检索**:自动从 `PubMed/Medline`、`arXiv`、`medRxiv`、`chemRxiv` 和 `bioRxiv` 搜索并获取论文元数据与全文记录。项目主要聚焦于生物医学与计算交叉领域(`Biomedicine + Computational Biology`)。
|
|
84
84
|
- **全文获取**:支持自动从 `PMC` 下载开放获取的 XML/Text 全文。对于预印本及其他没有 PMC 全文的文献,集成了额外的获取模块以下载 `原始 PDF`,并将 `Sci-Hub` 作为兜底来源。
|
|
85
|
+
- **预印本全文获取(免 PDF 解析)**:对于没有开放获取 PDF 的预印本,提供专用方法直接返回带章节标题的纯文本——`ArxivFetcher.fetch_full_text(arxiv_id)` 读取 ar5iv 渲染 HTML(arXiv LaTeX→HTML),`BioRxivFetcher.fetch_full_text(doi)` / `EuropePMCFullText.full_text_xml(doi)` 从 Europe PMC 读取 JATS 全文 XML(bioRxiv / medRxiv 及其它 DOI 收录预印本)。
|
|
85
86
|
- **结构化存储**:
|
|
86
87
|
- **元数据**:保存为结构清晰的详细 JSON 文件。
|
|
87
88
|
- **全文**:保存为多种格式,包括解析后的 JSON 和 Markdown,方便下游使用。其中 JSON 适合程序化分析,Markdown 更适合 LLM 理解与处理。
|
|
@@ -1440,6 +1441,31 @@ mineru_config.yaml mineru_export_config.yaml
|
|
|
1440
1441
|
|
|
1441
1442
|
> ⚠️ `针对上述预印本平台的模块目前基本已经开发完毕,后续只对相关功能进行维护和优化`, 测试细节与pubmed合并,详情见[Cases](./docs/Cases.md)
|
|
1442
1443
|
|
|
1444
|
+
#### 预印本全文获取(Python API)
|
|
1445
|
+
|
|
1446
|
+
对于没有开放获取 PDF 的预印本,各 fetcher 提供 `fetch_full_text()` 方法,免 PDF 解析直接返回带章节标题的纯文本:
|
|
1447
|
+
|
|
1448
|
+
```python
|
|
1449
|
+
from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher
|
|
1450
|
+
from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher
|
|
1451
|
+
from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
|
|
1452
|
+
|
|
1453
|
+
# arXiv → ar5iv 渲染 HTML(LaTeX → HTML)
|
|
1454
|
+
arxiv = ArxivFetcher(root_dir="./papers")
|
|
1455
|
+
text = arxiv.fetch_full_text("1706.03762") # 失败时返回 ""
|
|
1456
|
+
|
|
1457
|
+
# bioRxiv / medRxiv → Europe PMC fullTextXML
|
|
1458
|
+
biorxiv = BioRxivFetcher(root_dir="./papers", platform="biorxiv")
|
|
1459
|
+
text = biorxiv.fetch_full_text("10.1101/2023.06.22.546069")
|
|
1460
|
+
|
|
1461
|
+
# 任意 DOI 收录预印本 → 直接取 Europe PMC fullTextXML
|
|
1462
|
+
epmc = EuropePMCFullText()
|
|
1463
|
+
xml = epmc.full_text_xml("10.1101/2023.06.22.546069")
|
|
1464
|
+
epmc.close()
|
|
1465
|
+
```
|
|
1466
|
+
|
|
1467
|
+
三者失败时均返回空字符串 `""`,调用方可优雅回退到摘要。返回文本为带章节标题(`## Section`)的纯文本,可直接作为 LLM 输入。
|
|
1468
|
+
|
|
1443
1469
|
|
|
1444
1470
|
#### 1. 命令速查 (TL;DR)
|
|
1445
1471
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.6.1"
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import random
|
|
3
4
|
import sys
|
|
5
|
+
import threading
|
|
4
6
|
import time
|
|
5
7
|
import re
|
|
6
8
|
from datetime import datetime, timedelta
|
|
@@ -35,6 +37,22 @@ MED_RXIV_LAUNCH_DATE = datetime(2019, 6, 1)
|
|
|
35
37
|
|
|
36
38
|
DOI_RE = re.compile(r"^10\.\d{4,9}/[^\s]+$")
|
|
37
39
|
|
|
40
|
+
# bioRxiv/medRxiv return HTTP 429 for non-browser User-Agents on the .full-text
|
|
41
|
+
# route; a browser UA is required to fetch rendered full text.
|
|
42
|
+
_FULLTEXT_HEADERS = {
|
|
43
|
+
"User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
|
|
44
|
+
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
# Rate-limit / bot-wall hygiene for the .full-text route. bioRxiv/medRxiv sit
|
|
48
|
+
# behind a Cloudflare wall that 429s on rapid successive requests, so we (a)
|
|
49
|
+
# space requests apart, and (b) back off exponentially on 429/403/5xx rather
|
|
50
|
+
# than treating them like a missing article.
|
|
51
|
+
_FULLTEXT_MIN_INTERVAL = 2.0 # seconds between .full-text requests (shared across instances)
|
|
52
|
+
_FULLTEXT_BACKOFF_BASE = 3.0 # seconds; doubled per retry
|
|
53
|
+
_FULLTEXT_BACKOFF_CAP = 20.0 # seconds; hard ceiling on a single backoff sleep
|
|
54
|
+
_FULLTEXT_COOLDOWN = 30.0 # seconds to pause all requests after a 429/403
|
|
55
|
+
|
|
38
56
|
|
|
39
57
|
def _jats_xml_to_text(xml: str) -> str:
|
|
40
58
|
"""Convert JATS full-text XML to section-headed plain text."""
|
|
@@ -50,6 +68,35 @@ def _jats_xml_to_text(xml: str) -> str:
|
|
|
50
68
|
parts.append(text)
|
|
51
69
|
return "\n\n".join(parts).strip()
|
|
52
70
|
|
|
71
|
+
|
|
72
|
+
def _biorxiv_html_to_text(html: str) -> str:
|
|
73
|
+
"""Convert bioRxiv/medRxiv full-text HTML to section-headed plain text.
|
|
74
|
+
|
|
75
|
+
Walks the article body in document order, emitting headings as ``## ...``
|
|
76
|
+
and paragraphs as body text. Stops at the "References" section and skips
|
|
77
|
+
figure/table caption paragraphs.
|
|
78
|
+
"""
|
|
79
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
80
|
+
article = soup.find("div", class_="fulltext-view") or soup
|
|
81
|
+
parts: List[str] = []
|
|
82
|
+
for node in article.find_all(["h1", "h2", "h3", "p"]):
|
|
83
|
+
if node.name in ("h1", "h2", "h3"):
|
|
84
|
+
text = normalize_text(node.get_text(" ", strip=True))
|
|
85
|
+
if not text:
|
|
86
|
+
continue
|
|
87
|
+
if text.lower() == "references":
|
|
88
|
+
break
|
|
89
|
+
parts.append(f"\n## {text}")
|
|
90
|
+
else:
|
|
91
|
+
parent = node.find_parent("div")
|
|
92
|
+
classes = set(parent.get("class") or []) if parent is not None else set()
|
|
93
|
+
if classes & {"fig-caption", "table-caption"}:
|
|
94
|
+
continue
|
|
95
|
+
text = normalize_text(node.get_text(" ", strip=True))
|
|
96
|
+
if text:
|
|
97
|
+
parts.append(text)
|
|
98
|
+
return "\n\n".join(parts).strip()
|
|
99
|
+
|
|
53
100
|
PLATFORM_CONFIG = {
|
|
54
101
|
"biorxiv": {
|
|
55
102
|
"journal": "bioRxiv",
|
|
@@ -65,6 +112,12 @@ PLATFORM_CONFIG = {
|
|
|
65
112
|
|
|
66
113
|
|
|
67
114
|
class BioRxivFetcher:
|
|
115
|
+
# Shared across instances so sequential fetches (even from separate
|
|
116
|
+
# BioRxivFetcher objects) never fire back-to-back .full-text requests.
|
|
117
|
+
_fulltext_last_request = 0.0
|
|
118
|
+
_fulltext_cooldown_until = 0.0
|
|
119
|
+
_fulltext_lock = threading.Lock()
|
|
120
|
+
|
|
68
121
|
def __init__(
|
|
69
122
|
self,
|
|
70
123
|
root_dir: str,
|
|
@@ -88,11 +141,15 @@ class BioRxivFetcher:
|
|
|
88
141
|
"Accept": "application/json,text/html;q=0.9,*/*;q=0.8",
|
|
89
142
|
}
|
|
90
143
|
self._http_client: Optional[httpx.Client] = None
|
|
144
|
+
self._fulltext_client: Optional[httpx.Client] = None
|
|
91
145
|
|
|
92
146
|
def close(self) -> None:
|
|
93
147
|
if self._http_client is not None:
|
|
94
148
|
self._http_client.close()
|
|
95
149
|
self._http_client = None
|
|
150
|
+
if self._fulltext_client is not None:
|
|
151
|
+
self._fulltext_client.close()
|
|
152
|
+
self._fulltext_client = None
|
|
96
153
|
|
|
97
154
|
def __del__(self) -> None:
|
|
98
155
|
try:
|
|
@@ -398,12 +455,86 @@ class BioRxivFetcher:
|
|
|
398
455
|
return records
|
|
399
456
|
|
|
400
457
|
def fetch_full_text(self, doi: str) -> str:
|
|
401
|
-
"""Return full text for a bioRxiv/medRxiv preprint
|
|
402
|
-
from .europepmc_fetcher import EuropePMCFullText
|
|
458
|
+
"""Return full text for a bioRxiv/medRxiv preprint, or "" on failure.
|
|
403
459
|
|
|
404
|
-
|
|
460
|
+
Primary route: the preprint's own full-text HTML on biorxiv.org /
|
|
461
|
+
medrxiv.org. Fallback: Europe PMC fullTextXML (only present once the
|
|
462
|
+
preprint is published into PMC).
|
|
463
|
+
"""
|
|
464
|
+
doi = re.sub(r"v\d+$", "", normalize_text(doi))
|
|
405
465
|
if not doi:
|
|
406
466
|
return ""
|
|
467
|
+
text = self._fetch_full_text_html(doi)
|
|
468
|
+
if text:
|
|
469
|
+
return text
|
|
470
|
+
return self._fetch_full_text_europepmc(doi)
|
|
471
|
+
|
|
472
|
+
def _fulltext_http_client(self) -> httpx.Client:
|
|
473
|
+
if self._fulltext_client is None:
|
|
474
|
+
self._fulltext_client = httpx.Client(
|
|
475
|
+
headers=_FULLTEXT_HEADERS,
|
|
476
|
+
timeout=self.request_timeout,
|
|
477
|
+
follow_redirects=True,
|
|
478
|
+
trust_env=False,
|
|
479
|
+
)
|
|
480
|
+
return self._fulltext_client
|
|
481
|
+
|
|
482
|
+
@staticmethod
|
|
483
|
+
def _throttle_fulltext() -> None:
|
|
484
|
+
"""Space .full-text requests so they don't trip the Cloudflare wall.
|
|
485
|
+
|
|
486
|
+
Waits out both the minimum inter-request gap and any active post-429
|
|
487
|
+
cooldown, so a rate-limited fetch pauses the whole batch rather than
|
|
488
|
+
hammering the wall request after request.
|
|
489
|
+
"""
|
|
490
|
+
with BioRxivFetcher._fulltext_lock:
|
|
491
|
+
now = time.monotonic()
|
|
492
|
+
wait = max(
|
|
493
|
+
_FULLTEXT_MIN_INTERVAL - (now - BioRxivFetcher._fulltext_last_request),
|
|
494
|
+
BioRxivFetcher._fulltext_cooldown_until - now,
|
|
495
|
+
)
|
|
496
|
+
if wait > 0:
|
|
497
|
+
time.sleep(wait)
|
|
498
|
+
BioRxivFetcher._fulltext_last_request = time.monotonic()
|
|
499
|
+
|
|
500
|
+
@staticmethod
|
|
501
|
+
def _mark_fulltext_throttled() -> None:
|
|
502
|
+
with BioRxivFetcher._fulltext_lock:
|
|
503
|
+
BioRxivFetcher._fulltext_cooldown_until = time.monotonic() + _FULLTEXT_COOLDOWN
|
|
504
|
+
|
|
505
|
+
@staticmethod
|
|
506
|
+
def _backoff_seconds(attempt: int) -> float:
|
|
507
|
+
base = _FULLTEXT_BACKOFF_BASE * (2 ** attempt)
|
|
508
|
+
return min(_FULLTEXT_BACKOFF_CAP, base) + random.uniform(0, 1)
|
|
509
|
+
|
|
510
|
+
def _fetch_full_text_html(self, doi: str) -> str:
|
|
511
|
+
url = f"{self.landing_base}/{doi}.full-text"
|
|
512
|
+
client = self._fulltext_http_client()
|
|
513
|
+
for attempt in range(self.max_retries):
|
|
514
|
+
self._throttle_fulltext()
|
|
515
|
+
try:
|
|
516
|
+
response = client.get(url)
|
|
517
|
+
except Exception:
|
|
518
|
+
if attempt + 1 < self.max_retries:
|
|
519
|
+
time.sleep(self._backoff_seconds(attempt))
|
|
520
|
+
continue
|
|
521
|
+
status = response.status_code
|
|
522
|
+
if status == 200:
|
|
523
|
+
return _biorxiv_html_to_text(response.text)
|
|
524
|
+
if status == 404:
|
|
525
|
+
# No full-text page for this DOI; don't burn retries.
|
|
526
|
+
return ""
|
|
527
|
+
# 429 / 403 / 5xx: transient (rate limit / bot wall / server error).
|
|
528
|
+
# Mark a global cooldown so the rest of the batch pauses, then back
|
|
529
|
+
# off and retry.
|
|
530
|
+
self._mark_fulltext_throttled()
|
|
531
|
+
if attempt + 1 < self.max_retries:
|
|
532
|
+
time.sleep(self._backoff_seconds(attempt))
|
|
533
|
+
return ""
|
|
534
|
+
|
|
535
|
+
def _fetch_full_text_europepmc(self, doi: str) -> str:
|
|
536
|
+
from .europepmc_fetcher import EuropePMCFullText
|
|
537
|
+
|
|
407
538
|
fetcher = EuropePMCFullText(
|
|
408
539
|
request_timeout=self.request_timeout,
|
|
409
540
|
max_retries=self.max_retries,
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "0.6.0"
|
|
@@ -1,47 +0,0 @@
|
|
|
1
|
-
from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher, _html_to_text
|
|
2
|
-
|
|
3
|
-
HTML = """<html><body>
|
|
4
|
-
<h1 class="ltx_title">Title</h1>
|
|
5
|
-
<section class="ltx_section"><h2 class="ltx_title">Introduction</h2>
|
|
6
|
-
<p class="ltx_p">First paragraph.</p><p class="ltx_p">Second paragraph.</p></section>
|
|
7
|
-
<section class="ltx_section"><h2 class="ltx_title">Results</h2><p class="ltx_p">Result text.</p></section>
|
|
8
|
-
</body></html>"""
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
def test_html_to_text_extracts_sections():
|
|
12
|
-
text = _html_to_text(HTML)
|
|
13
|
-
assert "## Introduction" in text
|
|
14
|
-
assert "First paragraph." in text
|
|
15
|
-
assert "## Results" in text
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
def test_fetch_full_text_returns_body():
|
|
19
|
-
fetcher = ArxivFetcher(root_dir="/tmp")
|
|
20
|
-
|
|
21
|
-
class FakeResp:
|
|
22
|
-
status_code = 200
|
|
23
|
-
text = HTML
|
|
24
|
-
|
|
25
|
-
class FakeClient:
|
|
26
|
-
def get(self, url, follow_redirects=True):
|
|
27
|
-
assert url == "https://ar5iv.labs.arxiv.org/html/2301.00001"
|
|
28
|
-
return FakeResp()
|
|
29
|
-
|
|
30
|
-
fetcher._http_client = FakeClient()
|
|
31
|
-
text = fetcher.fetch_full_text("2301.00001v2")
|
|
32
|
-
assert "First paragraph." in text
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def test_fetch_full_text_404_returns_empty():
|
|
36
|
-
fetcher = ArxivFetcher(root_dir="/tmp")
|
|
37
|
-
|
|
38
|
-
class FakeResp:
|
|
39
|
-
status_code = 404
|
|
40
|
-
text = ""
|
|
41
|
-
|
|
42
|
-
class FakeClient:
|
|
43
|
-
def get(self, url, follow_redirects=True):
|
|
44
|
-
return FakeResp()
|
|
45
|
-
|
|
46
|
-
fetcher._http_client = FakeClient()
|
|
47
|
-
assert fetcher.fetch_full_text("2301.00001") == ""
|
|
@@ -1,101 +0,0 @@
|
|
|
1
|
-
import types
|
|
2
|
-
|
|
3
|
-
from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher, _jats_xml_to_text
|
|
4
|
-
from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
|
|
5
|
-
|
|
6
|
-
XML = """<article><body><sec><title>Abstract</title><p>Abs text.</p></sec>
|
|
7
|
-
<sec><title>Results</title><p>Result one.</p><p>Result two.</p></sec></body></article>"""
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
def test_jats_xml_to_text():
|
|
11
|
-
text = _jats_xml_to_text(XML)
|
|
12
|
-
assert "## Abstract" in text
|
|
13
|
-
assert "Abs text." in text
|
|
14
|
-
assert "## Results" in text
|
|
15
|
-
assert "Result two." in text
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
def test_full_text_xml_resolves_then_fetches():
|
|
19
|
-
ft = EuropePMCFullText()
|
|
20
|
-
|
|
21
|
-
def search_resp():
|
|
22
|
-
r = types.SimpleNamespace()
|
|
23
|
-
r.raise_for_status = lambda: None
|
|
24
|
-
r.json = lambda: {"resultList": {"result": [{"pmcid": "PMC123"}]}}
|
|
25
|
-
return r
|
|
26
|
-
|
|
27
|
-
def xml_resp():
|
|
28
|
-
r = types.SimpleNamespace()
|
|
29
|
-
r.raise_for_status = lambda: None
|
|
30
|
-
r.text = XML
|
|
31
|
-
return r
|
|
32
|
-
|
|
33
|
-
class FakeClient:
|
|
34
|
-
def __init__(self):
|
|
35
|
-
self.urls = []
|
|
36
|
-
|
|
37
|
-
def get(self, url, params=None):
|
|
38
|
-
self.urls.append(url)
|
|
39
|
-
if "fullTextXML" in url:
|
|
40
|
-
return xml_resp()
|
|
41
|
-
return search_resp()
|
|
42
|
-
|
|
43
|
-
def close(self):
|
|
44
|
-
pass
|
|
45
|
-
|
|
46
|
-
ft._client = FakeClient()
|
|
47
|
-
xml = ft.full_text_xml("10.1101/2026.01.01.123456")
|
|
48
|
-
assert "Results" in xml
|
|
49
|
-
assert any("PMC/PMC123/fullTextXML" in u for u in ft._client.urls)
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
def test_full_text_xml_no_result_empty():
|
|
53
|
-
ft = EuropePMCFullText()
|
|
54
|
-
r = types.SimpleNamespace()
|
|
55
|
-
r.raise_for_status = lambda: None
|
|
56
|
-
r.json = lambda: {"resultList": {"result": []}}
|
|
57
|
-
|
|
58
|
-
class FakeClient:
|
|
59
|
-
def get(self, url, params=None):
|
|
60
|
-
return r
|
|
61
|
-
|
|
62
|
-
def close(self):
|
|
63
|
-
pass
|
|
64
|
-
|
|
65
|
-
ft._client = FakeClient()
|
|
66
|
-
assert ft.full_text_xml("10.1101/2026.01.01.999999") == ""
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
def test_biorxiv_fetch_full_text(monkeypatch):
|
|
70
|
-
class FakeFT:
|
|
71
|
-
def __init__(self, request_timeout, max_retries):
|
|
72
|
-
pass
|
|
73
|
-
|
|
74
|
-
def full_text_xml(self, doi):
|
|
75
|
-
assert doi == "10.1101/2026.01.01.123456"
|
|
76
|
-
return XML
|
|
77
|
-
|
|
78
|
-
def close(self):
|
|
79
|
-
pass
|
|
80
|
-
|
|
81
|
-
monkeypatch.setattr("pyPaperFlow.preprint.europepmc_fetcher.EuropePMCFullText", FakeFT)
|
|
82
|
-
fetcher = BioRxivFetcher(root_dir="/tmp", platform="biorxiv")
|
|
83
|
-
text = fetcher.fetch_full_text("10.1101/2026.01.01.123456")
|
|
84
|
-
assert "## Results" in text
|
|
85
|
-
assert "Result one." in text
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
def test_biorxiv_fetch_full_text_empty(monkeypatch):
|
|
89
|
-
class FakeFT:
|
|
90
|
-
def __init__(self, request_timeout, max_retries):
|
|
91
|
-
pass
|
|
92
|
-
|
|
93
|
-
def full_text_xml(self, doi):
|
|
94
|
-
return ""
|
|
95
|
-
|
|
96
|
-
def close(self):
|
|
97
|
-
pass
|
|
98
|
-
|
|
99
|
-
monkeypatch.setattr("pyPaperFlow.preprint.europepmc_fetcher.EuropePMCFullText", FakeFT)
|
|
100
|
-
fetcher = BioRxivFetcher(root_dir="/tmp", platform="biorxiv")
|
|
101
|
-
assert fetcher.fetch_full_text("10.1101/2026.01.01.999999") == ""
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|