pyPaperFlow 0.6.0__tar.gz → 0.6.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/PKG-INFO +27 -1
  2. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/README.md +26 -0
  3. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/README_zh.md +26 -0
  4. pypaperflow-0.6.1/src/pyPaperFlow/__init__.py +1 -0
  5. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/biorxiv_fetcher.py +134 -3
  6. pypaperflow-0.6.0/src/pyPaperFlow/__init__.py +0 -1
  7. pypaperflow-0.6.0/tests/test_arxiv_fulltext.py +0 -47
  8. pypaperflow-0.6.0/tests/test_biorxiv_fulltext.py +0 -101
  9. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/.claude/settings.local.json +0 -0
  10. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/.github/workflows/docs.yml +0 -0
  11. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/.gitignore +0 -0
  12. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/LICENSE +0 -0
  13. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/mkdoc_site/index.md +0 -0
  14. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/mkdocs.yml +0 -0
  15. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/pyproject.toml +0 -0
  16. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/requirements-docs.txt +0 -0
  17. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/scripts/sync_docs.py +0 -0
  18. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/cli.py +0 -0
  19. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/cloak_fallback.py +0 -0
  20. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/cloak_pdf.py +0 -0
  21. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/github_export.py +0 -0
  22. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/mineru_parser.py +0 -0
  23. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/pdf_fetch.py +0 -0
  24. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/undetected_fallback.py +0 -0
  25. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/integrations/undetected_pdf.py +0 -0
  26. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/arxiv_fetcher.py +0 -0
  27. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/chemrxiv_fetcher.py +0 -0
  28. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/europepmc_fetcher.py +0 -0
  29. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/source_merge.py +0 -0
  30. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/source_models.py +0 -0
  31. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/preprint/source_utils.py +0 -0
  32. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/pubmed/__init__.py +0 -0
  33. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/pubmed/pubmed_fetcher.py +0 -0
  34. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/pubmed/pubmed_merger.py +0 -0
  35. {pypaperflow-0.6.0 → pypaperflow-0.6.1}/src/pyPaperFlow/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pyPaperFlow
3
- Version: 0.6.0
3
+ Version: 0.6.1
4
4
  Summary: Automated paper fetching and analysis platform.
5
5
  Project-URL: Homepage, https://github.com/MaybeBio/pyPaperFlow
6
6
  Project-URL: Issues, https://github.com/MaybeBio/pyPaperFlow/issues
@@ -112,6 +112,7 @@ This tool is designed to `complement rather than replace` reference management s
112
112
 
113
113
  - **Automated Retrieval from Multiple Sources**: Automatically search and retrieve paper metadata and full-text records from `PubMed/Medline, arXiv, medRxiv, chemRxiv and bioRxiv`. The repository focuses primarily on biomedical research and computational interdisciplinary fields (`Biomedicine + Computational Biology`).
114
114
  - **Full-Text Access**: Enable automatic downloading of open-access full texts in XML/Text format from `PMC`. For preprints and other publications without accessible PMC full texts, alternative acquisition modules are integrated to fetch `original PDFs`, with `Sci-Hub` set as the fallback provider.
115
+ - **Preprint Full-Text Fetch (no PDF parsing)**: For preprints without an OA PDF, dedicated methods return clean section-headed text directly — `ArxivFetcher.fetch_full_text(arxiv_id)` reads ar5iv-rendered HTML (arXiv LaTeX → HTML), and `BioRxivFetcher.fetch_full_text(doi)` / `EuropePMCFullText.full_text_xml(doi)` read JATS full-text XML from Europe PMC (bioRxiv / medRxiv and other DOI-indexed preprints).
115
116
  - **Structured Storage**:
116
117
  - **Metadata**: Preserved in well-structured detailed JSON files.
117
118
  - **Full Text**: Stored in multiple formats including parsed JSON and Markdown for versatile downstream usage — JSON for programmatic data analysis, and Markdown optimized for LLM comprehension and processing.
@@ -1450,6 +1451,31 @@ In theory, all DOI‑driven literature workflows can be standardised following t
1450
1451
 
1451
1452
  > Modules dedicated to the aforementioned preprint platforms are still under development and refinement. Preprint‑related subcommands are provided for testing purposes only. For detailed test cases, refer to [Cases](./docs/Cases.md)
1452
1453
 
1454
+ #### Preprint full-text fetch (Python API)
1455
+
1456
+ For preprints without an open-access PDF, each fetcher exposes a `fetch_full_text()` method that returns clean section-headed text without any PDF parsing:
1457
+
1458
+ ```python
1459
+ from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher
1460
+ from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher
1461
+ from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
1462
+
1463
+ # arXiv → ar5iv rendered HTML (LaTeX → HTML)
1464
+ arxiv = ArxivFetcher(root_dir="./papers")
1465
+ text = arxiv.fetch_full_text("1706.03762") # "" on failure
1466
+
1467
+ # bioRxiv / medRxiv → Europe PMC fullTextXML
1468
+ biorxiv = BioRxivFetcher(root_dir="./papers", platform="biorxiv")
1469
+ text = biorxiv.fetch_full_text("10.1101/2023.06.22.546069")
1470
+
1471
+ # Any DOI-indexed preprint → Europe PMC fullTextXML directly
1472
+ epmc = EuropePMCFullText()
1473
+ xml = epmc.full_text_xml("10.1101/2023.06.22.546069")
1474
+ epmc.close()
1475
+ ```
1476
+
1477
+ All three return an empty string `""` on failure, so callers can gracefully fall back to the abstract. The returned text is section-headed (`## Section`) plain text ready for LLM input.
1478
+
1453
1479
  ### 6. Critical Reading and Knowledge Graph Analysis: Downstream End‑Use
1454
1480
 
1455
1481
  Upon completing literature retrieval, parsing, and structured processing as outlined above, users obtain chapter‑organised Markdown files and structured JSON files, which serve as the fundamental inputs for subsequent critical reading and knowledge graph analysis.
@@ -83,6 +83,7 @@ This tool is designed to `complement rather than replace` reference management s
83
83
 
84
84
  - **Automated Retrieval from Multiple Sources**: Automatically search and retrieve paper metadata and full-text records from `PubMed/Medline, arXiv, medRxiv, chemRxiv and bioRxiv`. The repository focuses primarily on biomedical research and computational interdisciplinary fields (`Biomedicine + Computational Biology`).
85
85
  - **Full-Text Access**: Enable automatic downloading of open-access full texts in XML/Text format from `PMC`. For preprints and other publications without accessible PMC full texts, alternative acquisition modules are integrated to fetch `original PDFs`, with `Sci-Hub` set as the fallback provider.
86
+ - **Preprint Full-Text Fetch (no PDF parsing)**: For preprints without an OA PDF, dedicated methods return clean section-headed text directly — `ArxivFetcher.fetch_full_text(arxiv_id)` reads ar5iv-rendered HTML (arXiv LaTeX → HTML), and `BioRxivFetcher.fetch_full_text(doi)` / `EuropePMCFullText.full_text_xml(doi)` read JATS full-text XML from Europe PMC (bioRxiv / medRxiv and other DOI-indexed preprints).
86
87
  - **Structured Storage**:
87
88
  - **Metadata**: Preserved in well-structured detailed JSON files.
88
89
  - **Full Text**: Stored in multiple formats including parsed JSON and Markdown for versatile downstream usage — JSON for programmatic data analysis, and Markdown optimized for LLM comprehension and processing.
@@ -1421,6 +1422,31 @@ In theory, all DOI‑driven literature workflows can be standardised following t
1421
1422
 
1422
1423
  > Modules dedicated to the aforementioned preprint platforms are still under development and refinement. Preprint‑related subcommands are provided for testing purposes only. For detailed test cases, refer to [Cases](./docs/Cases.md)
1423
1424
 
1425
+ #### Preprint full-text fetch (Python API)
1426
+
1427
+ For preprints without an open-access PDF, each fetcher exposes a `fetch_full_text()` method that returns clean section-headed text without any PDF parsing:
1428
+
1429
+ ```python
1430
+ from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher
1431
+ from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher
1432
+ from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
1433
+
1434
+ # arXiv → ar5iv rendered HTML (LaTeX → HTML)
1435
+ arxiv = ArxivFetcher(root_dir="./papers")
1436
+ text = arxiv.fetch_full_text("1706.03762") # "" on failure
1437
+
1438
+ # bioRxiv / medRxiv → Europe PMC fullTextXML
1439
+ biorxiv = BioRxivFetcher(root_dir="./papers", platform="biorxiv")
1440
+ text = biorxiv.fetch_full_text("10.1101/2023.06.22.546069")
1441
+
1442
+ # Any DOI-indexed preprint → Europe PMC fullTextXML directly
1443
+ epmc = EuropePMCFullText()
1444
+ xml = epmc.full_text_xml("10.1101/2023.06.22.546069")
1445
+ epmc.close()
1446
+ ```
1447
+
1448
+ All three return an empty string `""` on failure, so callers can gracefully fall back to the abstract. The returned text is section-headed (`## Section`) plain text ready for LLM input.
1449
+
1424
1450
  ### 6. Critical Reading and Knowledge Graph Analysis: Downstream End‑Use
1425
1451
 
1426
1452
  Upon completing literature retrieval, parsing, and structured processing as outlined above, users obtain chapter‑organised Markdown files and structured JSON files, which serve as the fundamental inputs for subsequent critical reading and knowledge graph analysis.
@@ -82,6 +82,7 @@
82
82
 
83
83
  - **多来源自动检索**:自动从 `PubMed/Medline`、`arXiv`、`medRxiv`、`chemRxiv` 和 `bioRxiv` 搜索并获取论文元数据与全文记录。项目主要聚焦于生物医学与计算交叉领域(`Biomedicine + Computational Biology`)。
84
84
  - **全文获取**:支持自动从 `PMC` 下载开放获取的 XML/Text 全文。对于预印本及其他没有 PMC 全文的文献,集成了额外的获取模块以下载 `原始 PDF`,并将 `Sci-Hub` 作为兜底来源。
85
+ - **预印本全文获取(免 PDF 解析)**:对于没有开放获取 PDF 的预印本,提供专用方法直接返回带章节标题的纯文本——`ArxivFetcher.fetch_full_text(arxiv_id)` 读取 ar5iv 渲染 HTML(arXiv LaTeX→HTML),`BioRxivFetcher.fetch_full_text(doi)` / `EuropePMCFullText.full_text_xml(doi)` 从 Europe PMC 读取 JATS 全文 XML(bioRxiv / medRxiv 及其它 DOI 收录预印本)。
85
86
  - **结构化存储**:
86
87
  - **元数据**:保存为结构清晰的详细 JSON 文件。
87
88
  - **全文**:保存为多种格式,包括解析后的 JSON 和 Markdown,方便下游使用。其中 JSON 适合程序化分析,Markdown 更适合 LLM 理解与处理。
@@ -1440,6 +1441,31 @@ mineru_config.yaml mineru_export_config.yaml
1440
1441
 
1441
1442
  > ⚠️ `针对上述预印本平台的模块目前基本已经开发完毕,后续只对相关功能进行维护和优化`, 测试细节与pubmed合并,详情见[Cases](./docs/Cases.md)
1442
1443
 
1444
+ #### 预印本全文获取(Python API)
1445
+
1446
+ 对于没有开放获取 PDF 的预印本,各 fetcher 提供 `fetch_full_text()` 方法,免 PDF 解析直接返回带章节标题的纯文本:
1447
+
1448
+ ```python
1449
+ from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher
1450
+ from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher
1451
+ from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
1452
+
1453
+ # arXiv → ar5iv 渲染 HTML(LaTeX → HTML)
1454
+ arxiv = ArxivFetcher(root_dir="./papers")
1455
+ text = arxiv.fetch_full_text("1706.03762") # 失败时返回 ""
1456
+
1457
+ # bioRxiv / medRxiv → Europe PMC fullTextXML
1458
+ biorxiv = BioRxivFetcher(root_dir="./papers", platform="biorxiv")
1459
+ text = biorxiv.fetch_full_text("10.1101/2023.06.22.546069")
1460
+
1461
+ # 任意 DOI 收录预印本 → 直接取 Europe PMC fullTextXML
1462
+ epmc = EuropePMCFullText()
1463
+ xml = epmc.full_text_xml("10.1101/2023.06.22.546069")
1464
+ epmc.close()
1465
+ ```
1466
+
1467
+ 三者失败时均返回空字符串 `""`,调用方可优雅回退到摘要。返回文本为带章节标题(`## Section`)的纯文本,可直接作为 LLM 输入。
1468
+
1443
1469
 
1444
1470
  #### 1. 命令速查 (TL;DR)
1445
1471
 
@@ -0,0 +1 @@
1
+ __version__ = "0.6.1"
@@ -1,6 +1,8 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import random
3
4
  import sys
5
+ import threading
4
6
  import time
5
7
  import re
6
8
  from datetime import datetime, timedelta
@@ -35,6 +37,22 @@ MED_RXIV_LAUNCH_DATE = datetime(2019, 6, 1)
35
37
 
36
38
  DOI_RE = re.compile(r"^10\.\d{4,9}/[^\s]+$")
37
39
 
40
+ # bioRxiv/medRxiv return HTTP 429 for non-browser User-Agents on the .full-text
41
+ # route; a browser UA is required to fetch rendered full text.
42
+ _FULLTEXT_HEADERS = {
43
+ "User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
44
+ "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
45
+ }
46
+
47
+ # Rate-limit / bot-wall hygiene for the .full-text route. bioRxiv/medRxiv sit
48
+ # behind a Cloudflare wall that 429s on rapid successive requests, so we (a)
49
+ # space requests apart, and (b) back off exponentially on 429/403/5xx rather
50
+ # than treating them like a missing article.
51
+ _FULLTEXT_MIN_INTERVAL = 2.0 # seconds between .full-text requests (shared across instances)
52
+ _FULLTEXT_BACKOFF_BASE = 3.0 # seconds; doubled per retry
53
+ _FULLTEXT_BACKOFF_CAP = 20.0 # seconds; hard ceiling on a single backoff sleep
54
+ _FULLTEXT_COOLDOWN = 30.0 # seconds to pause all requests after a 429/403
55
+
38
56
 
39
57
  def _jats_xml_to_text(xml: str) -> str:
40
58
  """Convert JATS full-text XML to section-headed plain text."""
@@ -50,6 +68,35 @@ def _jats_xml_to_text(xml: str) -> str:
50
68
  parts.append(text)
51
69
  return "\n\n".join(parts).strip()
52
70
 
71
+
72
+ def _biorxiv_html_to_text(html: str) -> str:
73
+ """Convert bioRxiv/medRxiv full-text HTML to section-headed plain text.
74
+
75
+ Walks the article body in document order, emitting headings as ``## ...``
76
+ and paragraphs as body text. Stops at the "References" section and skips
77
+ figure/table caption paragraphs.
78
+ """
79
+ soup = BeautifulSoup(html, "html.parser")
80
+ article = soup.find("div", class_="fulltext-view") or soup
81
+ parts: List[str] = []
82
+ for node in article.find_all(["h1", "h2", "h3", "p"]):
83
+ if node.name in ("h1", "h2", "h3"):
84
+ text = normalize_text(node.get_text(" ", strip=True))
85
+ if not text:
86
+ continue
87
+ if text.lower() == "references":
88
+ break
89
+ parts.append(f"\n## {text}")
90
+ else:
91
+ parent = node.find_parent("div")
92
+ classes = set(parent.get("class") or []) if parent is not None else set()
93
+ if classes & {"fig-caption", "table-caption"}:
94
+ continue
95
+ text = normalize_text(node.get_text(" ", strip=True))
96
+ if text:
97
+ parts.append(text)
98
+ return "\n\n".join(parts).strip()
99
+
53
100
  PLATFORM_CONFIG = {
54
101
  "biorxiv": {
55
102
  "journal": "bioRxiv",
@@ -65,6 +112,12 @@ PLATFORM_CONFIG = {
65
112
 
66
113
 
67
114
  class BioRxivFetcher:
115
+ # Shared across instances so sequential fetches (even from separate
116
+ # BioRxivFetcher objects) never fire back-to-back .full-text requests.
117
+ _fulltext_last_request = 0.0
118
+ _fulltext_cooldown_until = 0.0
119
+ _fulltext_lock = threading.Lock()
120
+
68
121
  def __init__(
69
122
  self,
70
123
  root_dir: str,
@@ -88,11 +141,15 @@ class BioRxivFetcher:
88
141
  "Accept": "application/json,text/html;q=0.9,*/*;q=0.8",
89
142
  }
90
143
  self._http_client: Optional[httpx.Client] = None
144
+ self._fulltext_client: Optional[httpx.Client] = None
91
145
 
92
146
  def close(self) -> None:
93
147
  if self._http_client is not None:
94
148
  self._http_client.close()
95
149
  self._http_client = None
150
+ if self._fulltext_client is not None:
151
+ self._fulltext_client.close()
152
+ self._fulltext_client = None
96
153
 
97
154
  def __del__(self) -> None:
98
155
  try:
@@ -398,12 +455,86 @@ class BioRxivFetcher:
398
455
  return records
399
456
 
400
457
  def fetch_full_text(self, doi: str) -> str:
401
- """Return full text for a bioRxiv/medRxiv preprint via Europe PMC, or ""."""
402
- from .europepmc_fetcher import EuropePMCFullText
458
+ """Return full text for a bioRxiv/medRxiv preprint, or "" on failure.
403
459
 
404
- doi = normalize_text(doi)
460
+ Primary route: the preprint's own full-text HTML on biorxiv.org /
461
+ medrxiv.org. Fallback: Europe PMC fullTextXML (only present once the
462
+ preprint is published into PMC).
463
+ """
464
+ doi = re.sub(r"v\d+$", "", normalize_text(doi))
405
465
  if not doi:
406
466
  return ""
467
+ text = self._fetch_full_text_html(doi)
468
+ if text:
469
+ return text
470
+ return self._fetch_full_text_europepmc(doi)
471
+
472
+ def _fulltext_http_client(self) -> httpx.Client:
473
+ if self._fulltext_client is None:
474
+ self._fulltext_client = httpx.Client(
475
+ headers=_FULLTEXT_HEADERS,
476
+ timeout=self.request_timeout,
477
+ follow_redirects=True,
478
+ trust_env=False,
479
+ )
480
+ return self._fulltext_client
481
+
482
+ @staticmethod
483
+ def _throttle_fulltext() -> None:
484
+ """Space .full-text requests so they don't trip the Cloudflare wall.
485
+
486
+ Waits out both the minimum inter-request gap and any active post-429
487
+ cooldown, so a rate-limited fetch pauses the whole batch rather than
488
+ hammering the wall request after request.
489
+ """
490
+ with BioRxivFetcher._fulltext_lock:
491
+ now = time.monotonic()
492
+ wait = max(
493
+ _FULLTEXT_MIN_INTERVAL - (now - BioRxivFetcher._fulltext_last_request),
494
+ BioRxivFetcher._fulltext_cooldown_until - now,
495
+ )
496
+ if wait > 0:
497
+ time.sleep(wait)
498
+ BioRxivFetcher._fulltext_last_request = time.monotonic()
499
+
500
+ @staticmethod
501
+ def _mark_fulltext_throttled() -> None:
502
+ with BioRxivFetcher._fulltext_lock:
503
+ BioRxivFetcher._fulltext_cooldown_until = time.monotonic() + _FULLTEXT_COOLDOWN
504
+
505
+ @staticmethod
506
+ def _backoff_seconds(attempt: int) -> float:
507
+ base = _FULLTEXT_BACKOFF_BASE * (2 ** attempt)
508
+ return min(_FULLTEXT_BACKOFF_CAP, base) + random.uniform(0, 1)
509
+
510
+ def _fetch_full_text_html(self, doi: str) -> str:
511
+ url = f"{self.landing_base}/{doi}.full-text"
512
+ client = self._fulltext_http_client()
513
+ for attempt in range(self.max_retries):
514
+ self._throttle_fulltext()
515
+ try:
516
+ response = client.get(url)
517
+ except Exception:
518
+ if attempt + 1 < self.max_retries:
519
+ time.sleep(self._backoff_seconds(attempt))
520
+ continue
521
+ status = response.status_code
522
+ if status == 200:
523
+ return _biorxiv_html_to_text(response.text)
524
+ if status == 404:
525
+ # No full-text page for this DOI; don't burn retries.
526
+ return ""
527
+ # 429 / 403 / 5xx: transient (rate limit / bot wall / server error).
528
+ # Mark a global cooldown so the rest of the batch pauses, then back
529
+ # off and retry.
530
+ self._mark_fulltext_throttled()
531
+ if attempt + 1 < self.max_retries:
532
+ time.sleep(self._backoff_seconds(attempt))
533
+ return ""
534
+
535
+ def _fetch_full_text_europepmc(self, doi: str) -> str:
536
+ from .europepmc_fetcher import EuropePMCFullText
537
+
407
538
  fetcher = EuropePMCFullText(
408
539
  request_timeout=self.request_timeout,
409
540
  max_retries=self.max_retries,
@@ -1 +0,0 @@
1
- __version__ = "0.6.0"
@@ -1,47 +0,0 @@
1
- from pyPaperFlow.preprint.arxiv_fetcher import ArxivFetcher, _html_to_text
2
-
3
- HTML = """<html><body>
4
- <h1 class="ltx_title">Title</h1>
5
- <section class="ltx_section"><h2 class="ltx_title">Introduction</h2>
6
- <p class="ltx_p">First paragraph.</p><p class="ltx_p">Second paragraph.</p></section>
7
- <section class="ltx_section"><h2 class="ltx_title">Results</h2><p class="ltx_p">Result text.</p></section>
8
- </body></html>"""
9
-
10
-
11
- def test_html_to_text_extracts_sections():
12
- text = _html_to_text(HTML)
13
- assert "## Introduction" in text
14
- assert "First paragraph." in text
15
- assert "## Results" in text
16
-
17
-
18
- def test_fetch_full_text_returns_body():
19
- fetcher = ArxivFetcher(root_dir="/tmp")
20
-
21
- class FakeResp:
22
- status_code = 200
23
- text = HTML
24
-
25
- class FakeClient:
26
- def get(self, url, follow_redirects=True):
27
- assert url == "https://ar5iv.labs.arxiv.org/html/2301.00001"
28
- return FakeResp()
29
-
30
- fetcher._http_client = FakeClient()
31
- text = fetcher.fetch_full_text("2301.00001v2")
32
- assert "First paragraph." in text
33
-
34
-
35
- def test_fetch_full_text_404_returns_empty():
36
- fetcher = ArxivFetcher(root_dir="/tmp")
37
-
38
- class FakeResp:
39
- status_code = 404
40
- text = ""
41
-
42
- class FakeClient:
43
- def get(self, url, follow_redirects=True):
44
- return FakeResp()
45
-
46
- fetcher._http_client = FakeClient()
47
- assert fetcher.fetch_full_text("2301.00001") == ""
@@ -1,101 +0,0 @@
1
- import types
2
-
3
- from pyPaperFlow.preprint.biorxiv_fetcher import BioRxivFetcher, _jats_xml_to_text
4
- from pyPaperFlow.preprint.europepmc_fetcher import EuropePMCFullText
5
-
6
- XML = """<article><body><sec><title>Abstract</title><p>Abs text.</p></sec>
7
- <sec><title>Results</title><p>Result one.</p><p>Result two.</p></sec></body></article>"""
8
-
9
-
10
- def test_jats_xml_to_text():
11
- text = _jats_xml_to_text(XML)
12
- assert "## Abstract" in text
13
- assert "Abs text." in text
14
- assert "## Results" in text
15
- assert "Result two." in text
16
-
17
-
18
- def test_full_text_xml_resolves_then_fetches():
19
- ft = EuropePMCFullText()
20
-
21
- def search_resp():
22
- r = types.SimpleNamespace()
23
- r.raise_for_status = lambda: None
24
- r.json = lambda: {"resultList": {"result": [{"pmcid": "PMC123"}]}}
25
- return r
26
-
27
- def xml_resp():
28
- r = types.SimpleNamespace()
29
- r.raise_for_status = lambda: None
30
- r.text = XML
31
- return r
32
-
33
- class FakeClient:
34
- def __init__(self):
35
- self.urls = []
36
-
37
- def get(self, url, params=None):
38
- self.urls.append(url)
39
- if "fullTextXML" in url:
40
- return xml_resp()
41
- return search_resp()
42
-
43
- def close(self):
44
- pass
45
-
46
- ft._client = FakeClient()
47
- xml = ft.full_text_xml("10.1101/2026.01.01.123456")
48
- assert "Results" in xml
49
- assert any("PMC/PMC123/fullTextXML" in u for u in ft._client.urls)
50
-
51
-
52
- def test_full_text_xml_no_result_empty():
53
- ft = EuropePMCFullText()
54
- r = types.SimpleNamespace()
55
- r.raise_for_status = lambda: None
56
- r.json = lambda: {"resultList": {"result": []}}
57
-
58
- class FakeClient:
59
- def get(self, url, params=None):
60
- return r
61
-
62
- def close(self):
63
- pass
64
-
65
- ft._client = FakeClient()
66
- assert ft.full_text_xml("10.1101/2026.01.01.999999") == ""
67
-
68
-
69
- def test_biorxiv_fetch_full_text(monkeypatch):
70
- class FakeFT:
71
- def __init__(self, request_timeout, max_retries):
72
- pass
73
-
74
- def full_text_xml(self, doi):
75
- assert doi == "10.1101/2026.01.01.123456"
76
- return XML
77
-
78
- def close(self):
79
- pass
80
-
81
- monkeypatch.setattr("pyPaperFlow.preprint.europepmc_fetcher.EuropePMCFullText", FakeFT)
82
- fetcher = BioRxivFetcher(root_dir="/tmp", platform="biorxiv")
83
- text = fetcher.fetch_full_text("10.1101/2026.01.01.123456")
84
- assert "## Results" in text
85
- assert "Result one." in text
86
-
87
-
88
- def test_biorxiv_fetch_full_text_empty(monkeypatch):
89
- class FakeFT:
90
- def __init__(self, request_timeout, max_retries):
91
- pass
92
-
93
- def full_text_xml(self, doi):
94
- return ""
95
-
96
- def close(self):
97
- pass
98
-
99
- monkeypatch.setattr("pyPaperFlow.preprint.europepmc_fetcher.EuropePMCFullText", FakeFT)
100
- fetcher = BioRxivFetcher(root_dir="/tmp", platform="biorxiv")
101
- assert fetcher.fetch_full_text("10.1101/2026.01.01.999999") == ""
File without changes
File without changes
File without changes
File without changes