paperstack-cli 0.3.0__py3-none-any.whl → 0.3.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
paperstack/arxiv.py ADDED
@@ -0,0 +1,125 @@
1
+ """Bounded arXiv discovery with source-backed metadata."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import urllib.parse
7
+ import xml.etree.ElementTree as ET
8
+ from datetime import UTC, datetime
9
+
10
+ from . import metadata
11
+
12
+ ATOM = "http://www.w3.org/2005/Atom"
13
+ ARXIV = "http://arxiv.org/schemas/atom"
14
+ NS = {"atom": ATOM, "arxiv": ARXIV}
15
+ MAX_RESULTS = 100
16
+ _CATEGORY = re.compile(r"^[a-z-]+(?:\.[A-Za-z-]+)?$")
17
+
18
+
19
+ def _date(value: str, *, end: bool = False) -> str:
20
+ try:
21
+ parsed = datetime.fromisoformat(value)
22
+ except ValueError as exc:
23
+ raise ValueError(f"invalid date {value!r}; use YYYY-MM-DD or ISO 8601") from exc
24
+ if len(value) == 10:
25
+ parsed = parsed.replace(hour=23 if end else 0, minute=59 if end else 0)
26
+ elif parsed.tzinfo is not None:
27
+ parsed = parsed.astimezone(UTC)
28
+ return parsed.strftime("%Y%m%d%H%M")
29
+
30
+
31
+ def _query(
32
+ query: str,
33
+ *,
34
+ categories: list[str] | None = None,
35
+ date_from: str | None = None,
36
+ date_to: str | None = None,
37
+ ) -> str:
38
+ parts = []
39
+ if query.strip():
40
+ parts.append(f"({query.strip()})")
41
+ if categories:
42
+ invalid = [item for item in categories if not _CATEGORY.fullmatch(item)]
43
+ if invalid:
44
+ raise ValueError(f"invalid arXiv category: {invalid[0]}")
45
+ parts.append("(" + " OR ".join(f"cat:{item}" for item in categories) + ")")
46
+ if date_from or date_to:
47
+ start = _date(date_from) if date_from else "199107010000"
48
+ end = _date(date_to, end=True) if date_to else datetime.now(UTC).strftime("%Y%m%d%H%M")
49
+ if start > end:
50
+ raise ValueError("date-from must not be after date-to")
51
+ parts.append(f"submittedDate:[{start}+TO+{end}]")
52
+ if not parts:
53
+ raise ValueError("arXiv search needs a query, category, or date range")
54
+ return " AND ".join(parts)
55
+
56
+
57
+ def _text(entry: ET.Element, name: str) -> str | None:
58
+ value = entry.findtext(name, None, NS)
59
+ return " ".join(value.split()) if value else None
60
+
61
+
62
+ def _entry(entry: ET.Element) -> dict:
63
+ raw_id = _text(entry, "atom:id") or ""
64
+ versioned = raw_id.rsplit("/abs/", 1)[-1]
65
+ arxiv_id = re.sub(r"v\d+$", "", versioned)
66
+ links = {node.get("title") or node.get("rel"): node.get("href") for node in entry.findall("atom:link", NS)}
67
+ categories = [node.get("term") for node in entry.findall("atom:category", NS) if node.get("term")]
68
+ primary = entry.find("arxiv:primary_category", NS)
69
+ authors = [_text(node, "atom:name") for node in entry.findall("atom:author", NS)]
70
+ return {
71
+ "id": arxiv_id,
72
+ "versioned_id": versioned,
73
+ "title": _text(entry, "atom:title"),
74
+ "authors": [author for author in authors if author],
75
+ "abstract": _text(entry, "atom:summary"),
76
+ "categories": categories,
77
+ "primary_category": primary.get("term") if primary is not None else (categories[0] if categories else None),
78
+ "published": _text(entry, "atom:published"),
79
+ "updated": _text(entry, "atom:updated"),
80
+ "comment": _text(entry, "arxiv:comment"),
81
+ "journal_ref": _text(entry, "arxiv:journal_ref"),
82
+ "doi": _text(entry, "arxiv:doi"),
83
+ "url": f"https://arxiv.org/abs/{versioned}",
84
+ "pdf_url": links.get("pdf") or f"https://arxiv.org/pdf/{versioned}",
85
+ }
86
+
87
+
88
+ def parse_feed(raw: bytes | str) -> list[dict]:
89
+ root = ET.fromstring(raw)
90
+ return [_entry(entry) for entry in root.findall("atom:entry", NS)]
91
+
92
+
93
+ def search(
94
+ query: str,
95
+ *,
96
+ categories: list[str] | None = None,
97
+ date_from: str | None = None,
98
+ date_to: str | None = None,
99
+ limit: int = 10,
100
+ sort: str = "relevance",
101
+ ) -> dict:
102
+ if not 1 <= limit <= MAX_RESULTS:
103
+ raise ValueError(f"limit must be between 1 and {MAX_RESULTS}")
104
+ if sort not in ("relevance", "date"):
105
+ raise ValueError("sort must be relevance or date")
106
+ built = _query(query, categories=categories, date_from=date_from, date_to=date_to)
107
+ params = urllib.parse.urlencode(
108
+ {
109
+ "search_query": built,
110
+ "max_results": limit,
111
+ "sortBy": "submittedDate" if sort == "date" else "relevance",
112
+ "sortOrder": "descending",
113
+ }
114
+ )
115
+ params = params.replace("%2BTO%2B", "+TO+")
116
+ url = f"https://export.arxiv.org/api/query?{params}"
117
+ return metadata._safe(
118
+ lambda: metadata._result(
119
+ "arxiv",
120
+ url,
121
+ {"query": built, "matches": parse_feed(metadata.request(url))},
122
+ ),
123
+ "arxiv",
124
+ url,
125
+ )
paperstack/citations.py CHANGED
@@ -3,13 +3,12 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import json
6
- import os
7
6
  import re
8
7
  import urllib.parse
9
8
  from datetime import UTC, datetime
10
9
  from pathlib import Path
11
10
 
12
- from . import metadata
11
+ from . import credentials, metadata
13
12
 
14
13
  S2_BATCH_API = "https://api.semanticscholar.org/graph/v1/paper/batch"
15
14
  BATCH_SIZE = 500
@@ -44,7 +43,7 @@ def fetch(arxiv_ids: list[str]) -> dict[str, int]:
44
43
  """Fetch citation counts in aligned Semantic Scholar batches."""
45
44
  counts: dict[str, int] = {}
46
45
  headers = {"Content-Type": "application/json"}
47
- if api_key := os.environ.get("SEMANTIC_SCHOLAR_API_KEY"):
46
+ if api_key := credentials.get(credentials.SEMANTIC_SCHOLAR_API_KEY):
48
47
  headers["x-api-key"] = api_key
49
48
 
50
49
  for start in range(0, len(arxiv_ids), BATCH_SIZE):