paperstack-cli 0.3.0__py3-none-any.whl → 0.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- paperstack/arxiv.py +125 -0
- paperstack/citations.py +2 -3
- paperstack/cli.py +263 -37
- paperstack/content/arxiv_pdf.py +2 -2
- paperstack/corpora.py +27 -3
- paperstack/credentials.py +147 -0
- paperstack/dblp_build.py +3 -1
- paperstack/dblp_index.py +28 -54
- paperstack/entrypoint.py +9 -1
- paperstack/metadata.py +20 -8
- paperstack/semantic_scholar.py +175 -0
- paperstack/viewer.py +38 -10
- paperstack/viewer_assets/vendor/marked.LICENSE +36 -0
- {paperstack_cli-0.3.0.dist-info → paperstack_cli-0.3.2.dist-info}/METADATA +50 -7
- paperstack_cli-0.3.2.dist-info/RECORD +31 -0
- paperstack_cli-0.3.0.dist-info/RECORD +0 -27
- {paperstack_cli-0.3.0.dist-info → paperstack_cli-0.3.2.dist-info}/WHEEL +0 -0
- {paperstack_cli-0.3.0.dist-info → paperstack_cli-0.3.2.dist-info}/entry_points.txt +0 -0
- {paperstack_cli-0.3.0.dist-info → paperstack_cli-0.3.2.dist-info}/licenses/LICENSE +0 -0
paperstack/arxiv.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Bounded arXiv discovery with source-backed metadata."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import urllib.parse
|
|
7
|
+
import xml.etree.ElementTree as ET
|
|
8
|
+
from datetime import UTC, datetime
|
|
9
|
+
|
|
10
|
+
from . import metadata
|
|
11
|
+
|
|
12
|
+
ATOM = "http://www.w3.org/2005/Atom"
|
|
13
|
+
ARXIV = "http://arxiv.org/schemas/atom"
|
|
14
|
+
NS = {"atom": ATOM, "arxiv": ARXIV}
|
|
15
|
+
MAX_RESULTS = 100
|
|
16
|
+
_CATEGORY = re.compile(r"^[a-z-]+(?:\.[A-Za-z-]+)?$")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _date(value: str, *, end: bool = False) -> str:
|
|
20
|
+
try:
|
|
21
|
+
parsed = datetime.fromisoformat(value)
|
|
22
|
+
except ValueError as exc:
|
|
23
|
+
raise ValueError(f"invalid date {value!r}; use YYYY-MM-DD or ISO 8601") from exc
|
|
24
|
+
if len(value) == 10:
|
|
25
|
+
parsed = parsed.replace(hour=23 if end else 0, minute=59 if end else 0)
|
|
26
|
+
elif parsed.tzinfo is not None:
|
|
27
|
+
parsed = parsed.astimezone(UTC)
|
|
28
|
+
return parsed.strftime("%Y%m%d%H%M")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _query(
|
|
32
|
+
query: str,
|
|
33
|
+
*,
|
|
34
|
+
categories: list[str] | None = None,
|
|
35
|
+
date_from: str | None = None,
|
|
36
|
+
date_to: str | None = None,
|
|
37
|
+
) -> str:
|
|
38
|
+
parts = []
|
|
39
|
+
if query.strip():
|
|
40
|
+
parts.append(f"({query.strip()})")
|
|
41
|
+
if categories:
|
|
42
|
+
invalid = [item for item in categories if not _CATEGORY.fullmatch(item)]
|
|
43
|
+
if invalid:
|
|
44
|
+
raise ValueError(f"invalid arXiv category: {invalid[0]}")
|
|
45
|
+
parts.append("(" + " OR ".join(f"cat:{item}" for item in categories) + ")")
|
|
46
|
+
if date_from or date_to:
|
|
47
|
+
start = _date(date_from) if date_from else "199107010000"
|
|
48
|
+
end = _date(date_to, end=True) if date_to else datetime.now(UTC).strftime("%Y%m%d%H%M")
|
|
49
|
+
if start > end:
|
|
50
|
+
raise ValueError("date-from must not be after date-to")
|
|
51
|
+
parts.append(f"submittedDate:[{start}+TO+{end}]")
|
|
52
|
+
if not parts:
|
|
53
|
+
raise ValueError("arXiv search needs a query, category, or date range")
|
|
54
|
+
return " AND ".join(parts)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _text(entry: ET.Element, name: str) -> str | None:
|
|
58
|
+
value = entry.findtext(name, None, NS)
|
|
59
|
+
return " ".join(value.split()) if value else None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _entry(entry: ET.Element) -> dict:
|
|
63
|
+
raw_id = _text(entry, "atom:id") or ""
|
|
64
|
+
versioned = raw_id.rsplit("/abs/", 1)[-1]
|
|
65
|
+
arxiv_id = re.sub(r"v\d+$", "", versioned)
|
|
66
|
+
links = {node.get("title") or node.get("rel"): node.get("href") for node in entry.findall("atom:link", NS)}
|
|
67
|
+
categories = [node.get("term") for node in entry.findall("atom:category", NS) if node.get("term")]
|
|
68
|
+
primary = entry.find("arxiv:primary_category", NS)
|
|
69
|
+
authors = [_text(node, "atom:name") for node in entry.findall("atom:author", NS)]
|
|
70
|
+
return {
|
|
71
|
+
"id": arxiv_id,
|
|
72
|
+
"versioned_id": versioned,
|
|
73
|
+
"title": _text(entry, "atom:title"),
|
|
74
|
+
"authors": [author for author in authors if author],
|
|
75
|
+
"abstract": _text(entry, "atom:summary"),
|
|
76
|
+
"categories": categories,
|
|
77
|
+
"primary_category": primary.get("term") if primary is not None else (categories[0] if categories else None),
|
|
78
|
+
"published": _text(entry, "atom:published"),
|
|
79
|
+
"updated": _text(entry, "atom:updated"),
|
|
80
|
+
"comment": _text(entry, "arxiv:comment"),
|
|
81
|
+
"journal_ref": _text(entry, "arxiv:journal_ref"),
|
|
82
|
+
"doi": _text(entry, "arxiv:doi"),
|
|
83
|
+
"url": f"https://arxiv.org/abs/{versioned}",
|
|
84
|
+
"pdf_url": links.get("pdf") or f"https://arxiv.org/pdf/{versioned}",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def parse_feed(raw: bytes | str) -> list[dict]:
|
|
89
|
+
root = ET.fromstring(raw)
|
|
90
|
+
return [_entry(entry) for entry in root.findall("atom:entry", NS)]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def search(
|
|
94
|
+
query: str,
|
|
95
|
+
*,
|
|
96
|
+
categories: list[str] | None = None,
|
|
97
|
+
date_from: str | None = None,
|
|
98
|
+
date_to: str | None = None,
|
|
99
|
+
limit: int = 10,
|
|
100
|
+
sort: str = "relevance",
|
|
101
|
+
) -> dict:
|
|
102
|
+
if not 1 <= limit <= MAX_RESULTS:
|
|
103
|
+
raise ValueError(f"limit must be between 1 and {MAX_RESULTS}")
|
|
104
|
+
if sort not in ("relevance", "date"):
|
|
105
|
+
raise ValueError("sort must be relevance or date")
|
|
106
|
+
built = _query(query, categories=categories, date_from=date_from, date_to=date_to)
|
|
107
|
+
params = urllib.parse.urlencode(
|
|
108
|
+
{
|
|
109
|
+
"search_query": built,
|
|
110
|
+
"max_results": limit,
|
|
111
|
+
"sortBy": "submittedDate" if sort == "date" else "relevance",
|
|
112
|
+
"sortOrder": "descending",
|
|
113
|
+
}
|
|
114
|
+
)
|
|
115
|
+
params = params.replace("%2BTO%2B", "+TO+")
|
|
116
|
+
url = f"https://export.arxiv.org/api/query?{params}"
|
|
117
|
+
return metadata._safe(
|
|
118
|
+
lambda: metadata._result(
|
|
119
|
+
"arxiv",
|
|
120
|
+
url,
|
|
121
|
+
{"query": built, "matches": parse_feed(metadata.request(url))},
|
|
122
|
+
),
|
|
123
|
+
"arxiv",
|
|
124
|
+
url,
|
|
125
|
+
)
|
paperstack/citations.py
CHANGED
|
@@ -3,13 +3,12 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import json
|
|
6
|
-
import os
|
|
7
6
|
import re
|
|
8
7
|
import urllib.parse
|
|
9
8
|
from datetime import UTC, datetime
|
|
10
9
|
from pathlib import Path
|
|
11
10
|
|
|
12
|
-
from . import metadata
|
|
11
|
+
from . import credentials, metadata
|
|
13
12
|
|
|
14
13
|
S2_BATCH_API = "https://api.semanticscholar.org/graph/v1/paper/batch"
|
|
15
14
|
BATCH_SIZE = 500
|
|
@@ -44,7 +43,7 @@ def fetch(arxiv_ids: list[str]) -> dict[str, int]:
|
|
|
44
43
|
"""Fetch citation counts in aligned Semantic Scholar batches."""
|
|
45
44
|
counts: dict[str, int] = {}
|
|
46
45
|
headers = {"Content-Type": "application/json"}
|
|
47
|
-
if api_key :=
|
|
46
|
+
if api_key := credentials.get(credentials.SEMANTIC_SCHOLAR_API_KEY):
|
|
48
47
|
headers["x-api-key"] = api_key
|
|
49
48
|
|
|
50
49
|
for start in range(0, len(arxiv_ids), BATCH_SIZE):
|