paperstack-cli 0.4.0__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/PKG-INFO +6 -1
  2. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/README.md +5 -0
  3. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/cli.py +55 -17
  4. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/arxiv_pdf.py +3 -1
  5. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/arxiv_source.py +3 -1
  6. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/metadata.py +194 -45
  7. paperstack_cli-0.4.2/src/paperstack/tls.py +15 -0
  8. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/viewer.py +3 -2
  9. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/.gitignore +0 -0
  10. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/LICENSE +0 -0
  11. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/pyproject.toml +0 -0
  12. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/app.js +0 -0
  13. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/entry.html +0 -0
  14. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/favicon.svg +0 -0
  15. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/index.html +0 -0
  16. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/style.css +0 -0
  17. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/vendor/marked.LICENSE +0 -0
  18. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/vendor/marked.min.js +0 -0
  19. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/__init__.py +0 -0
  20. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/arxiv.py +0 -0
  21. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/citations.py +0 -0
  22. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/__init__.py +0 -0
  23. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/vendor/latexpand +0 -0
  24. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/vendor/latexpand.LICENSE +0 -0
  25. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/corpora.py +0 -0
  26. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/credentials.py +0 -0
  27. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/dblp_build.py +0 -0
  28. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/dblp_catalog.py +0 -0
  29. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/dblp_index.py +0 -0
  30. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/entry_types.py +0 -0
  31. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/entrypoint.py +0 -0
  32. {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/semantic_scholar.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: paperstack-cli
3
- Version: 0.4.0
3
+ Version: 0.4.2
4
4
  Summary: Review, inspect, and retrieve research sources from one CLI
5
5
  Project-URL: Repository, https://github.com/MilkClouds/paperstack
6
6
  Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
@@ -134,6 +134,7 @@ These commands use external source records and do not select a citation or make
134
134
 
135
135
  ```bash
136
136
  paperstack paper search "Attention Is All You Need" --source dblp
137
+ paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
137
138
  paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
138
139
  paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
139
140
  paperstack paper metadata arxiv:2106.09685
@@ -151,6 +152,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
151
152
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
152
153
  of a work should be cited.
153
154
 
155
+ OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
156
+ filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
157
+ public invitation, venue, decision, and status fields rather than treating it as a universal field.
158
+
154
159
  ## Build a viewer
155
160
 
156
161
  ```bash
@@ -112,6 +112,7 @@ These commands use external source records and do not select a citation or make
112
112
 
113
113
  ```bash
114
114
  paperstack paper search "Attention Is All You Need" --source dblp
115
+ paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
115
116
  paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
116
117
  paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
117
118
  paperstack paper metadata arxiv:2106.09685
@@ -129,6 +130,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
129
130
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
130
131
  of a work should be cited.
131
132
 
133
+ OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
134
+ filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
135
+ public invitation, venue, decision, and status fields rather than treating it as a universal field.
136
+
132
137
  ## Build a viewer
133
138
 
134
139
  ```bash
@@ -925,12 +925,42 @@ def _run_paper(a: argparse.Namespace) -> int:
925
925
  metadata.print_results(results, json_output=a.json)
926
926
  return 0 if any(item["status"] == "ok" for item in results) else 1
927
927
  if a.paper_cmd == "search":
928
+ semantic_filters = [
929
+ flag
930
+ for flag, selected in (
931
+ ("--offset", a.offset),
932
+ ("--year", a.year),
933
+ ("--field-of-study", a.fields_of_study),
934
+ ("--open-access", a.open_access),
935
+ )
936
+ if selected
937
+ ]
938
+ arxiv_filters = [
939
+ flag
940
+ for flag, selected in (
941
+ ("--category", a.categories),
942
+ ("--date-from", a.date_from),
943
+ ("--date-to", a.date_to),
944
+ ("--sort", a.sort != "relevance"),
945
+ )
946
+ if selected
947
+ ]
948
+ openreview_filters = [
949
+ flag
950
+ for flag, selected in (
951
+ ("--exact-title", a.exact_title),
952
+ ("--openreview-status", a.openreview_status),
953
+ )
954
+ if selected
955
+ ]
928
956
  try:
957
+ if a.source != "openreview" and openreview_filters:
958
+ die(f"--source openreview is required for {', '.join(openreview_filters)}")
929
959
  if a.source == "arxiv":
930
960
  from . import arxiv
931
961
 
932
- if a.year or a.fields_of_study or a.open_access or a.offset:
933
- die("--year, --field-of-study, --open-access, and --offset require --source semantic-scholar")
962
+ if semantic_filters:
963
+ die(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
934
964
  result = arxiv.search(
935
965
  a.query,
936
966
  categories=a.categories,
@@ -942,8 +972,8 @@ def _run_paper(a: argparse.Namespace) -> int:
942
972
  elif a.source == "semantic-scholar":
943
973
  from . import semantic_scholar
944
974
 
945
- if a.categories or a.date_from or a.date_to or a.sort != "relevance":
946
- die("--category, --date-from, --date-to, and --sort require --source arxiv")
975
+ if arxiv_filters:
976
+ die(f"--source arxiv is required for {', '.join(arxiv_filters)}")
947
977
  result = semantic_scholar.search(
948
978
  a.query,
949
979
  limit=a.limit,
@@ -953,19 +983,21 @@ def _run_paper(a: argparse.Namespace) -> int:
953
983
  open_access=a.open_access,
954
984
  )
955
985
  else:
956
- if (
957
- a.categories
958
- or a.date_from
959
- or a.date_to
960
- or a.limit != 10
961
- or a.offset
962
- or a.sort != "relevance"
963
- or a.year
964
- or a.fields_of_study
965
- or a.open_access
966
- ):
967
- die("search filters require --source arxiv or --source semantic-scholar")
968
- result = metadata.search(a.source, a.query, local_only=offline)
986
+ requirements = []
987
+ if semantic_filters:
988
+ requirements.append(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
989
+ if arxiv_filters:
990
+ requirements.append(f"--source arxiv is required for {', '.join(arxiv_filters)}")
991
+ if requirements:
992
+ die("; ".join(requirements))
993
+ result = metadata.search(
994
+ a.source,
995
+ a.query,
996
+ limit=a.limit,
997
+ local_only=offline,
998
+ exact_title=a.exact_title,
999
+ openreview_status=a.openreview_status,
1000
+ )
969
1001
  except credentials.CredentialsError as exc:
970
1002
  die(f"configuration failed: {exc}")
971
1003
  except RuntimeError as exc:
@@ -1178,6 +1210,12 @@ Use `paperstack review ...` to find or read an authored critical judgment.""",
1178
1210
  s.add_argument("--date-from", help="earliest arXiv submission date")
1179
1211
  s.add_argument("--date-to", help="latest arXiv submission date")
1180
1212
  s.add_argument("--sort", choices=("relevance", "date"), default="relevance")
1213
+ s.add_argument("--exact-title", action="store_true", help="require a normalized exact OpenReview title")
1214
+ s.add_argument(
1215
+ "--openreview-status",
1216
+ choices=("submission", "accepted", "withdrawn"),
1217
+ help="filter OpenReview forum records by inferred status",
1218
+ )
1181
1219
  _output(s)
1182
1220
  _offline(s)
1183
1221
  for command in ("authors", "citations", "references"):
@@ -16,6 +16,8 @@ import urllib.error
16
16
  import urllib.request
17
17
  from pathlib import Path
18
18
 
19
+ from ..tls import arxiv_context
20
+
19
21
  _CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
20
22
  CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
21
23
  CONVERTER = "pdf-inspector"
@@ -27,7 +29,7 @@ def _fetch(url: str, timeout: int = 60) -> bytes | None:
27
29
  req = urllib.request.Request(url, headers={"User-Agent": "paperstack/1.0 (+arxiv pdf fetch)"})
28
30
  for attempt in range(3):
29
31
  try:
30
- with urllib.request.urlopen(req, timeout=timeout) as resp:
32
+ with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
31
33
  return resp.read()
32
34
  except urllib.error.HTTPError as e:
33
35
  if e.code == 429:
@@ -22,6 +22,8 @@ import urllib.request
22
22
  import zlib
23
23
  from pathlib import Path
24
24
 
25
+ from ..tls import arxiv_context
26
+
25
27
  _CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
26
28
  CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
27
29
 
@@ -43,7 +45,7 @@ def _fetch_bytes(url: str, timeout: int = 60) -> bytes | None:
43
45
  )
44
46
  for attempt in range(3):
45
47
  try:
46
- with urllib.request.urlopen(req, timeout=timeout) as resp:
48
+ with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
47
49
  return resp.read()
48
50
  except urllib.error.HTTPError as e:
49
51
  if e.code == 429:
@@ -6,13 +6,21 @@ import json
6
6
  import os
7
7
  import re
8
8
  import time
9
+ import unicodedata
9
10
  import urllib.error
10
11
  import urllib.parse
11
12
  import urllib.request
12
13
  import xml.etree.ElementTree as ET
14
+ from contextlib import contextmanager
13
15
  from dataclasses import dataclass
16
+ from datetime import UTC, datetime
17
+ from email.utils import parsedate_to_datetime
18
+ from pathlib import Path
19
+
20
+ from filelock import FileLock
14
21
 
15
22
  from . import credentials
23
+ from .tls import arxiv_context
16
24
 
17
25
  ARXIV_NS = {"atom": "http://www.w3.org/2005/Atom", "arxiv": "http://arxiv.org/schemas/atom"}
18
26
  SOURCES = ("semantic_scholar", "dblp", "crossref", "openreview", "acl_anthology", "arxiv")
@@ -45,6 +53,59 @@ class PaperRef:
45
53
 
46
54
 
47
55
  _last_request: dict[str, float] = {}
56
+ _OPENREVIEW_HOSTS = {"api.openreview.net", "api2.openreview.net"}
57
+
58
+
59
+ @contextmanager
60
+ def _request_slot(key: str):
61
+ if key != "openreview":
62
+ yield None
63
+ return
64
+ base = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
65
+ root = base / "paperstack" / "http"
66
+ try:
67
+ root.mkdir(parents=True, exist_ok=True, mode=0o700)
68
+ lock = FileLock(root / "openreview.lock", timeout=120)
69
+ lock.acquire()
70
+ except OSError:
71
+ yield None
72
+ return
73
+ try:
74
+ yield root / "openreview.timestamp"
75
+ finally:
76
+ lock.release()
77
+
78
+
79
+ def _shared_elapsed(path: Path | None) -> float:
80
+ if path is None:
81
+ return float("inf")
82
+ try:
83
+ return max(0.0, time.time() - path.stat().st_mtime)
84
+ except OSError:
85
+ return float("inf")
86
+
87
+
88
+ def _mark_request(path: Path | None) -> None:
89
+ if path is not None:
90
+ try:
91
+ path.touch()
92
+ except OSError:
93
+ pass
94
+
95
+
96
+ def _retry_delay(exc: urllib.error.HTTPError, attempt: int) -> float:
97
+ value = exc.headers.get("Retry-After") if exc.headers else None
98
+ if value:
99
+ try:
100
+ delay = float(value)
101
+ except ValueError:
102
+ try:
103
+ delay = (parsedate_to_datetime(value) - datetime.now(UTC)).total_seconds()
104
+ except (TypeError, ValueError, OverflowError):
105
+ delay = -1
106
+ if delay >= 0:
107
+ return min(delay, 120)
108
+ return min(2**attempt, 30)
48
109
 
49
110
 
50
111
  def request(
@@ -56,34 +117,42 @@ def request(
56
117
  if params:
57
118
  url += "?" + urllib.parse.urlencode(params)
58
119
  host = urllib.parse.urlparse(url).netloc
120
+ request_key = "openreview" if host in _OPENREVIEW_HOSTS else host
59
121
  interval = 3.0 if "arxiv.org" in host else 1.1 if "dblp.org" in host else 0.5
60
122
  request_headers = {
61
123
  "User-Agent": "paperstack (+https://github.com/MilkClouds/paperstack)",
62
124
  **(headers or {}),
63
125
  }
64
126
  req = urllib.request.Request(url, headers=request_headers, data=data)
65
- for attempt in range(3):
66
- elapsed = time.monotonic() - _last_request.get(host, 0.0)
67
- if elapsed < interval:
68
- time.sleep(interval - elapsed)
69
- try:
70
- with urllib.request.urlopen(req, timeout=30) as response:
71
- _last_request[host] = time.monotonic()
72
- return response.read()
73
- except urllib.error.HTTPError as exc:
74
- _last_request[host] = time.monotonic()
75
- if exc.code != 429:
76
- raise
77
- if attempt == 2:
78
- has_api_key = any(name.lower() == "x-api-key" for name in request_headers)
79
- if host == "api.semanticscholar.org" and not has_api_key:
80
- message = (
81
- f"{exc.reason}; configure semantic-scholar.api-key with "
82
- "`paperstack config set semantic-scholar.api-key` for more reliable access"
83
- )
84
- raise urllib.error.HTTPError(exc.url, exc.code, message, exc.headers, exc.fp) from exc
85
- raise
86
- time.sleep(5 * (attempt + 1))
127
+ open_options = {"context": arxiv_context()} if "arxiv.org" in host else {}
128
+ with _request_slot(request_key) as shared_timestamp:
129
+ for attempt in range(3):
130
+ elapsed = min(
131
+ time.monotonic() - _last_request.get(request_key, 0.0),
132
+ _shared_elapsed(shared_timestamp),
133
+ )
134
+ if elapsed < interval:
135
+ time.sleep(interval - elapsed)
136
+ try:
137
+ with urllib.request.urlopen(req, timeout=30, **open_options) as response:
138
+ _last_request[request_key] = time.monotonic()
139
+ _mark_request(shared_timestamp)
140
+ return response.read()
141
+ except urllib.error.HTTPError as exc:
142
+ _last_request[request_key] = time.monotonic()
143
+ _mark_request(shared_timestamp)
144
+ if exc.code != 429:
145
+ raise
146
+ if attempt == 2:
147
+ has_api_key = any(name.lower() == "x-api-key" for name in request_headers)
148
+ if host == "api.semanticscholar.org" and not has_api_key:
149
+ message = (
150
+ f"{exc.reason}; configure semantic-scholar.api-key with "
151
+ "`paperstack config set semantic-scholar.api-key` for more reliable access"
152
+ )
153
+ raise urllib.error.HTTPError(exc.url, exc.code, message, exc.headers, exc.fp) from exc
154
+ raise
155
+ time.sleep(_retry_delay(exc, attempt))
87
156
  raise RuntimeError("unreachable request retry state")
88
157
 
89
158
 
@@ -305,11 +374,50 @@ def fetch_all(
305
374
  return results
306
375
 
307
376
 
308
- def search(source: str, query: str, *, local_only: bool = False) -> dict:
377
+ def _content_value(note: dict, name: str):
378
+ value = (note.get("content") or {}).get(name)
379
+ return value.get("value") if isinstance(value, dict) and "value" in value else value
380
+
381
+
382
+ def _normalized_title(value: object) -> str:
383
+ return re.sub(r"\W+", "", unicodedata.normalize("NFKC", str(value or "")).casefold())
384
+
385
+
386
+ def _openreview_status(note: dict) -> str:
387
+ venue = _content_value(note, "venue") or _content_value(note, "venueid") or _content_value(note, "venue_id")
388
+ fields = [*note.get("invitations", []), venue, _content_value(note, "decision"), _content_value(note, "status")]
389
+ text = " ".join(str(value) for value in fields if value).casefold()
390
+ if "withdraw" in text:
391
+ return "withdrawn"
392
+ if "reject" in text:
393
+ return "rejected"
394
+ if "accept" in text or venue and not any(word in str(venue).casefold() for word in ("submission", "submitted")):
395
+ return "accepted"
396
+ return "submission"
397
+
398
+
399
+ def search(
400
+ source: str,
401
+ query: str,
402
+ *,
403
+ limit: int = 10,
404
+ local_only: bool = False,
405
+ exact_title: bool = False,
406
+ openreview_status: str | None = None,
407
+ ) -> dict:
408
+ query = query.strip()
409
+ if not query:
410
+ raise ValueError("paper search query is required")
411
+ if not 1 <= limit <= 100:
412
+ raise ValueError("limit must be between 1 and 100")
413
+ if (exact_title or openreview_status) and source != "openreview":
414
+ raise ValueError("exact title and OpenReview status filters require the openreview source")
415
+ if openreview_status not in (None, "submission", "accepted", "withdrawn"):
416
+ raise ValueError("OpenReview status must be submission, accepted, or withdrawn")
309
417
  if source == "dblp":
310
418
  from . import dblp_index
311
419
 
312
- if hits := dblp_index.search(query):
420
+ if hits := dblp_index.search(query, limit=limit):
313
421
  return _result("dblp", str(dblp_index.index_path()), {"query": query, "matches": hits})
314
422
  if local_only:
315
423
  return {
@@ -319,19 +427,25 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
319
427
  "reason": "not found in local index",
320
428
  }
321
429
  url = "https://dblp.org/search/publ/api"
430
+ has_field_token = re.search(r"(?:^|\s)(?:author|title|venue|year|type|stream|toc):[^:]+:", query)
431
+ remote_query = query if has_field_token else re.sub(r":\s+", " ", query)
322
432
  return _safe(
323
- lambda: _result("dblp", url, _get_json(url, {"q": query, "format": "json", "h": 10})), "dblp", url
433
+ lambda: _result("dblp", url, _get_json(url, {"q": remote_query, "format": "json", "h": limit})),
434
+ "dblp",
435
+ url,
324
436
  )
325
437
  if source == "crossref":
326
438
  url = "https://api.crossref.org/works"
327
439
  return _safe(
328
- lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows": 10})), "crossref", url
440
+ lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows": limit})),
441
+ "crossref",
442
+ url,
329
443
  )
330
444
  if source == "arxiv":
331
445
  url = "https://export.arxiv.org/api/query"
332
446
 
333
447
  def arxiv_search() -> dict:
334
- root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results": 10}))
448
+ root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results": limit}))
335
449
  matches = [
336
450
  {
337
451
  "id": entry.findtext("atom:id", "", ARXIV_NS),
@@ -347,40 +461,75 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
347
461
  token = os.environ.get("OPENREVIEW_ACCESS_TOKEN")
348
462
  headers = {"Cookie": f"openreview.accessToken={token}"} if token else {}
349
463
  endpoints = (
350
- "https://api2.openreview.net/notes/search",
351
- "https://api.openreview.net/notes/search",
464
+ ("https://api2.openreview.net/notes/search", True),
465
+ ("https://api.openreview.net/notes/search", False),
352
466
  )
353
467
  matches = []
354
468
  errors = []
355
- for endpoint in endpoints:
356
- result = _safe(
357
- lambda endpoint=endpoint: _result(
469
+ for endpoint, is_v2 in endpoints:
470
+ endpoint_matches = []
471
+ local_filter = bool(openreview_status or (exact_title and not is_v2))
472
+ page_size = min(max(limit * 2, 20), 100) if local_filter else limit
473
+ params = {"limit": page_size, "source": "forum", "cache": "true"}
474
+ if exact_title and is_v2:
475
+ params.update({"term": query, "type": "exact", "content": "title"})
476
+ else:
477
+ params["query"] = query
478
+ for offset in range(0, 1000, page_size):
479
+ page_params = {**params, "offset": offset} if offset else params
480
+ result = _safe(
481
+ lambda endpoint=endpoint, page_params=page_params: _result(
482
+ "openreview",
483
+ endpoint,
484
+ _get_json(endpoint, page_params, headers),
485
+ ),
358
486
  "openreview",
359
487
  endpoint,
360
- _get_json(endpoint, {"query": query, "limit": 10, "source": "forum"}, headers),
361
- ),
362
- "openreview",
363
- endpoint,
364
- )
365
- if result["status"] == "ok":
366
- matches.extend(result.get("response", {}).get("notes", []))
367
- else:
368
- errors.append(result.get("error", endpoint))
488
+ )
489
+ if result["status"] != "ok":
490
+ errors.append(result.get("error", endpoint))
491
+ break
492
+ notes = result.get("response", {}).get("notes", [])
493
+ matches.extend(notes)
494
+ endpoint_matches.extend(notes)
495
+ if not local_filter or len(notes) < page_size:
496
+ break
497
+ eligible = endpoint_matches
498
+ if exact_title:
499
+ wanted = _normalized_title(query)
500
+ eligible = [
501
+ note for note in eligible if _normalized_title(_content_value(note, "title")) == wanted
502
+ ]
503
+ if openreview_status:
504
+ eligible = [note for note in eligible if _openreview_status(note) == openreview_status]
505
+ eligible_ids = {note.get("forum") or note.get("id") for note in eligible}
506
+ if len(eligible_ids) >= limit:
507
+ break
369
508
  if not matches and errors:
370
- return _result("openreview", endpoints[0], error="; ".join(errors))
509
+ return _result("openreview", endpoints[0][0], error="; ".join(errors))
371
510
  unique = {}
372
511
  for note in matches:
373
512
  unique[note.get("forum") or note.get("id") or json.dumps(note, sort_keys=True)] = note
513
+ selected = list(unique.values())
514
+ if exact_title:
515
+ wanted = _normalized_title(query)
516
+ selected = [note for note in selected if _normalized_title(_content_value(note, "title")) == wanted]
517
+ if openreview_status:
518
+ selected = [note for note in selected if _openreview_status(note) == openreview_status]
374
519
  return _result(
375
520
  "openreview",
376
- endpoints[0],
377
- {"query": query, "matches": list(unique.values()), "api_endpoints": list(endpoints)},
521
+ endpoints[0][0],
522
+ {
523
+ "query": query,
524
+ "matches": selected[:limit],
525
+ "api_endpoints": [endpoint for endpoint, _ in endpoints],
526
+ },
378
527
  )
379
528
  if source == "s2":
380
529
  url = "https://api.semanticscholar.org/graph/v1/paper/search"
381
530
  api_key = credentials.get(credentials.SEMANTIC_SCHOLAR_API_KEY)
382
531
  headers = {"x-api-key": api_key} if api_key else {}
383
- params = {"query": query, "limit": 10, "fields": S2_FIELDS}
532
+ params = {"query": query, "limit": limit, "fields": S2_FIELDS}
384
533
  return _safe(
385
534
  lambda: _result("semantic_scholar", url, _get_json(url, params, headers)), "semantic_scholar", url
386
535
  )
@@ -0,0 +1,15 @@
1
+ """TLS settings for hosts that reject some client handshakes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import functools
6
+ import ssl
7
+
8
+
9
+ @functools.cache
10
+ def arxiv_context() -> ssl.SSLContext:
11
+ # arXiv's Fastly edge answers uncached requests with 406 when the TLS 1.3
12
+ # ClientHello offers the X25519MLKEM768 hybrid key share (OpenSSL >= 3.5 default).
13
+ context = ssl.create_default_context()
14
+ context.set_ecdh_curve("X25519")
15
+ return context
@@ -94,11 +94,12 @@ def build(root: Path, output: Path) -> int:
94
94
  ):
95
95
  raise ValueError("viewer output would replace a broad path, the corpus, or authored entries")
96
96
  backup = destination.parent / f".{destination.name}.backup"
97
- if backup.exists() and not destination.exists():
97
+ if backup.exists():
98
98
  marker = backup / ".paperstack-viewer"
99
99
  if not backup.is_dir() or not marker.is_file() or marker.read_text(encoding="utf-8") != "1\n":
100
100
  raise ValueError(f"viewer backup is not owned by Paperstack: {backup}")
101
- os.replace(backup, destination)
101
+ if not destination.exists():
102
+ os.replace(backup, destination)
102
103
  if destination.exists() and not destination.is_dir():
103
104
  raise ValueError(f"viewer output is not a directory: {destination}")
104
105
  if destination.exists() and any(destination.iterdir()):
File without changes