paperstack-cli 0.4.0__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/PKG-INFO +6 -1
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/README.md +5 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/cli.py +55 -17
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/arxiv_pdf.py +3 -1
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/arxiv_source.py +3 -1
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/metadata.py +194 -45
- paperstack_cli-0.4.2/src/paperstack/tls.py +15 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/viewer.py +3 -2
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/.gitignore +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/LICENSE +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/pyproject.toml +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/app.js +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/entry.html +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/favicon.svg +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/index.html +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/site/style.css +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/vendor/marked.LICENSE +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/scripts/build/vendor/marked.min.js +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/__init__.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/arxiv.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/citations.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/__init__.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/vendor/latexpand +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/vendor/latexpand.LICENSE +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/corpora.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/credentials.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/dblp_build.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/dblp_catalog.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/dblp_index.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/entry_types.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/entrypoint.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/semantic_scholar.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: paperstack-cli
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Review, inspect, and retrieve research sources from one CLI
|
|
5
5
|
Project-URL: Repository, https://github.com/MilkClouds/paperstack
|
|
6
6
|
Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
|
|
@@ -134,6 +134,7 @@ These commands use external source records and do not select a citation or make
|
|
|
134
134
|
|
|
135
135
|
```bash
|
|
136
136
|
paperstack paper search "Attention Is All You Need" --source dblp
|
|
137
|
+
paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
|
|
137
138
|
paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
|
|
138
139
|
paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
|
|
139
140
|
paperstack paper metadata arxiv:2106.09685
|
|
@@ -151,6 +152,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
|
|
|
151
152
|
`--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
|
|
152
153
|
of a work should be cited.
|
|
153
154
|
|
|
155
|
+
OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
|
|
156
|
+
filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
|
|
157
|
+
public invitation, venue, decision, and status fields rather than treating it as a universal field.
|
|
158
|
+
|
|
154
159
|
## Build a viewer
|
|
155
160
|
|
|
156
161
|
```bash
|
|
@@ -112,6 +112,7 @@ These commands use external source records and do not select a citation or make
|
|
|
112
112
|
|
|
113
113
|
```bash
|
|
114
114
|
paperstack paper search "Attention Is All You Need" --source dblp
|
|
115
|
+
paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
|
|
115
116
|
paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
|
|
116
117
|
paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
|
|
117
118
|
paperstack paper metadata arxiv:2106.09685
|
|
@@ -129,6 +130,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
|
|
|
129
130
|
`--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
|
|
130
131
|
of a work should be cited.
|
|
131
132
|
|
|
133
|
+
OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
|
|
134
|
+
filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
|
|
135
|
+
public invitation, venue, decision, and status fields rather than treating it as a universal field.
|
|
136
|
+
|
|
132
137
|
## Build a viewer
|
|
133
138
|
|
|
134
139
|
```bash
|
|
@@ -925,12 +925,42 @@ def _run_paper(a: argparse.Namespace) -> int:
|
|
|
925
925
|
metadata.print_results(results, json_output=a.json)
|
|
926
926
|
return 0 if any(item["status"] == "ok" for item in results) else 1
|
|
927
927
|
if a.paper_cmd == "search":
|
|
928
|
+
semantic_filters = [
|
|
929
|
+
flag
|
|
930
|
+
for flag, selected in (
|
|
931
|
+
("--offset", a.offset),
|
|
932
|
+
("--year", a.year),
|
|
933
|
+
("--field-of-study", a.fields_of_study),
|
|
934
|
+
("--open-access", a.open_access),
|
|
935
|
+
)
|
|
936
|
+
if selected
|
|
937
|
+
]
|
|
938
|
+
arxiv_filters = [
|
|
939
|
+
flag
|
|
940
|
+
for flag, selected in (
|
|
941
|
+
("--category", a.categories),
|
|
942
|
+
("--date-from", a.date_from),
|
|
943
|
+
("--date-to", a.date_to),
|
|
944
|
+
("--sort", a.sort != "relevance"),
|
|
945
|
+
)
|
|
946
|
+
if selected
|
|
947
|
+
]
|
|
948
|
+
openreview_filters = [
|
|
949
|
+
flag
|
|
950
|
+
for flag, selected in (
|
|
951
|
+
("--exact-title", a.exact_title),
|
|
952
|
+
("--openreview-status", a.openreview_status),
|
|
953
|
+
)
|
|
954
|
+
if selected
|
|
955
|
+
]
|
|
928
956
|
try:
|
|
957
|
+
if a.source != "openreview" and openreview_filters:
|
|
958
|
+
die(f"--source openreview is required for {', '.join(openreview_filters)}")
|
|
929
959
|
if a.source == "arxiv":
|
|
930
960
|
from . import arxiv
|
|
931
961
|
|
|
932
|
-
if
|
|
933
|
-
die("--
|
|
962
|
+
if semantic_filters:
|
|
963
|
+
die(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
|
|
934
964
|
result = arxiv.search(
|
|
935
965
|
a.query,
|
|
936
966
|
categories=a.categories,
|
|
@@ -942,8 +972,8 @@ def _run_paper(a: argparse.Namespace) -> int:
|
|
|
942
972
|
elif a.source == "semantic-scholar":
|
|
943
973
|
from . import semantic_scholar
|
|
944
974
|
|
|
945
|
-
if
|
|
946
|
-
die("--
|
|
975
|
+
if arxiv_filters:
|
|
976
|
+
die(f"--source arxiv is required for {', '.join(arxiv_filters)}")
|
|
947
977
|
result = semantic_scholar.search(
|
|
948
978
|
a.query,
|
|
949
979
|
limit=a.limit,
|
|
@@ -953,19 +983,21 @@ def _run_paper(a: argparse.Namespace) -> int:
|
|
|
953
983
|
open_access=a.open_access,
|
|
954
984
|
)
|
|
955
985
|
else:
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
986
|
+
requirements = []
|
|
987
|
+
if semantic_filters:
|
|
988
|
+
requirements.append(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
|
|
989
|
+
if arxiv_filters:
|
|
990
|
+
requirements.append(f"--source arxiv is required for {', '.join(arxiv_filters)}")
|
|
991
|
+
if requirements:
|
|
992
|
+
die("; ".join(requirements))
|
|
993
|
+
result = metadata.search(
|
|
994
|
+
a.source,
|
|
995
|
+
a.query,
|
|
996
|
+
limit=a.limit,
|
|
997
|
+
local_only=offline,
|
|
998
|
+
exact_title=a.exact_title,
|
|
999
|
+
openreview_status=a.openreview_status,
|
|
1000
|
+
)
|
|
969
1001
|
except credentials.CredentialsError as exc:
|
|
970
1002
|
die(f"configuration failed: {exc}")
|
|
971
1003
|
except RuntimeError as exc:
|
|
@@ -1178,6 +1210,12 @@ Use `paperstack review ...` to find or read an authored critical judgment.""",
|
|
|
1178
1210
|
s.add_argument("--date-from", help="earliest arXiv submission date")
|
|
1179
1211
|
s.add_argument("--date-to", help="latest arXiv submission date")
|
|
1180
1212
|
s.add_argument("--sort", choices=("relevance", "date"), default="relevance")
|
|
1213
|
+
s.add_argument("--exact-title", action="store_true", help="require a normalized exact OpenReview title")
|
|
1214
|
+
s.add_argument(
|
|
1215
|
+
"--openreview-status",
|
|
1216
|
+
choices=("submission", "accepted", "withdrawn"),
|
|
1217
|
+
help="filter OpenReview forum records by inferred status",
|
|
1218
|
+
)
|
|
1181
1219
|
_output(s)
|
|
1182
1220
|
_offline(s)
|
|
1183
1221
|
for command in ("authors", "citations", "references"):
|
|
@@ -16,6 +16,8 @@ import urllib.error
|
|
|
16
16
|
import urllib.request
|
|
17
17
|
from pathlib import Path
|
|
18
18
|
|
|
19
|
+
from ..tls import arxiv_context
|
|
20
|
+
|
|
19
21
|
_CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
20
22
|
CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
|
|
21
23
|
CONVERTER = "pdf-inspector"
|
|
@@ -27,7 +29,7 @@ def _fetch(url: str, timeout: int = 60) -> bytes | None:
|
|
|
27
29
|
req = urllib.request.Request(url, headers={"User-Agent": "paperstack/1.0 (+arxiv pdf fetch)"})
|
|
28
30
|
for attempt in range(3):
|
|
29
31
|
try:
|
|
30
|
-
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
32
|
+
with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
|
|
31
33
|
return resp.read()
|
|
32
34
|
except urllib.error.HTTPError as e:
|
|
33
35
|
if e.code == 429:
|
|
@@ -22,6 +22,8 @@ import urllib.request
|
|
|
22
22
|
import zlib
|
|
23
23
|
from pathlib import Path
|
|
24
24
|
|
|
25
|
+
from ..tls import arxiv_context
|
|
26
|
+
|
|
25
27
|
_CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
26
28
|
CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
|
|
27
29
|
|
|
@@ -43,7 +45,7 @@ def _fetch_bytes(url: str, timeout: int = 60) -> bytes | None:
|
|
|
43
45
|
)
|
|
44
46
|
for attempt in range(3):
|
|
45
47
|
try:
|
|
46
|
-
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
48
|
+
with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
|
|
47
49
|
return resp.read()
|
|
48
50
|
except urllib.error.HTTPError as e:
|
|
49
51
|
if e.code == 429:
|
|
@@ -6,13 +6,21 @@ import json
|
|
|
6
6
|
import os
|
|
7
7
|
import re
|
|
8
8
|
import time
|
|
9
|
+
import unicodedata
|
|
9
10
|
import urllib.error
|
|
10
11
|
import urllib.parse
|
|
11
12
|
import urllib.request
|
|
12
13
|
import xml.etree.ElementTree as ET
|
|
14
|
+
from contextlib import contextmanager
|
|
13
15
|
from dataclasses import dataclass
|
|
16
|
+
from datetime import UTC, datetime
|
|
17
|
+
from email.utils import parsedate_to_datetime
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from filelock import FileLock
|
|
14
21
|
|
|
15
22
|
from . import credentials
|
|
23
|
+
from .tls import arxiv_context
|
|
16
24
|
|
|
17
25
|
ARXIV_NS = {"atom": "http://www.w3.org/2005/Atom", "arxiv": "http://arxiv.org/schemas/atom"}
|
|
18
26
|
SOURCES = ("semantic_scholar", "dblp", "crossref", "openreview", "acl_anthology", "arxiv")
|
|
@@ -45,6 +53,59 @@ class PaperRef:
|
|
|
45
53
|
|
|
46
54
|
|
|
47
55
|
_last_request: dict[str, float] = {}
|
|
56
|
+
_OPENREVIEW_HOSTS = {"api.openreview.net", "api2.openreview.net"}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@contextmanager
|
|
60
|
+
def _request_slot(key: str):
|
|
61
|
+
if key != "openreview":
|
|
62
|
+
yield None
|
|
63
|
+
return
|
|
64
|
+
base = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
65
|
+
root = base / "paperstack" / "http"
|
|
66
|
+
try:
|
|
67
|
+
root.mkdir(parents=True, exist_ok=True, mode=0o700)
|
|
68
|
+
lock = FileLock(root / "openreview.lock", timeout=120)
|
|
69
|
+
lock.acquire()
|
|
70
|
+
except OSError:
|
|
71
|
+
yield None
|
|
72
|
+
return
|
|
73
|
+
try:
|
|
74
|
+
yield root / "openreview.timestamp"
|
|
75
|
+
finally:
|
|
76
|
+
lock.release()
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _shared_elapsed(path: Path | None) -> float:
|
|
80
|
+
if path is None:
|
|
81
|
+
return float("inf")
|
|
82
|
+
try:
|
|
83
|
+
return max(0.0, time.time() - path.stat().st_mtime)
|
|
84
|
+
except OSError:
|
|
85
|
+
return float("inf")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _mark_request(path: Path | None) -> None:
|
|
89
|
+
if path is not None:
|
|
90
|
+
try:
|
|
91
|
+
path.touch()
|
|
92
|
+
except OSError:
|
|
93
|
+
pass
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _retry_delay(exc: urllib.error.HTTPError, attempt: int) -> float:
|
|
97
|
+
value = exc.headers.get("Retry-After") if exc.headers else None
|
|
98
|
+
if value:
|
|
99
|
+
try:
|
|
100
|
+
delay = float(value)
|
|
101
|
+
except ValueError:
|
|
102
|
+
try:
|
|
103
|
+
delay = (parsedate_to_datetime(value) - datetime.now(UTC)).total_seconds()
|
|
104
|
+
except (TypeError, ValueError, OverflowError):
|
|
105
|
+
delay = -1
|
|
106
|
+
if delay >= 0:
|
|
107
|
+
return min(delay, 120)
|
|
108
|
+
return min(2**attempt, 30)
|
|
48
109
|
|
|
49
110
|
|
|
50
111
|
def request(
|
|
@@ -56,34 +117,42 @@ def request(
|
|
|
56
117
|
if params:
|
|
57
118
|
url += "?" + urllib.parse.urlencode(params)
|
|
58
119
|
host = urllib.parse.urlparse(url).netloc
|
|
120
|
+
request_key = "openreview" if host in _OPENREVIEW_HOSTS else host
|
|
59
121
|
interval = 3.0 if "arxiv.org" in host else 1.1 if "dblp.org" in host else 0.5
|
|
60
122
|
request_headers = {
|
|
61
123
|
"User-Agent": "paperstack (+https://github.com/MilkClouds/paperstack)",
|
|
62
124
|
**(headers or {}),
|
|
63
125
|
}
|
|
64
126
|
req = urllib.request.Request(url, headers=request_headers, data=data)
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
127
|
+
open_options = {"context": arxiv_context()} if "arxiv.org" in host else {}
|
|
128
|
+
with _request_slot(request_key) as shared_timestamp:
|
|
129
|
+
for attempt in range(3):
|
|
130
|
+
elapsed = min(
|
|
131
|
+
time.monotonic() - _last_request.get(request_key, 0.0),
|
|
132
|
+
_shared_elapsed(shared_timestamp),
|
|
133
|
+
)
|
|
134
|
+
if elapsed < interval:
|
|
135
|
+
time.sleep(interval - elapsed)
|
|
136
|
+
try:
|
|
137
|
+
with urllib.request.urlopen(req, timeout=30, **open_options) as response:
|
|
138
|
+
_last_request[request_key] = time.monotonic()
|
|
139
|
+
_mark_request(shared_timestamp)
|
|
140
|
+
return response.read()
|
|
141
|
+
except urllib.error.HTTPError as exc:
|
|
142
|
+
_last_request[request_key] = time.monotonic()
|
|
143
|
+
_mark_request(shared_timestamp)
|
|
144
|
+
if exc.code != 429:
|
|
145
|
+
raise
|
|
146
|
+
if attempt == 2:
|
|
147
|
+
has_api_key = any(name.lower() == "x-api-key" for name in request_headers)
|
|
148
|
+
if host == "api.semanticscholar.org" and not has_api_key:
|
|
149
|
+
message = (
|
|
150
|
+
f"{exc.reason}; configure semantic-scholar.api-key with "
|
|
151
|
+
"`paperstack config set semantic-scholar.api-key` for more reliable access"
|
|
152
|
+
)
|
|
153
|
+
raise urllib.error.HTTPError(exc.url, exc.code, message, exc.headers, exc.fp) from exc
|
|
154
|
+
raise
|
|
155
|
+
time.sleep(_retry_delay(exc, attempt))
|
|
87
156
|
raise RuntimeError("unreachable request retry state")
|
|
88
157
|
|
|
89
158
|
|
|
@@ -305,11 +374,50 @@ def fetch_all(
|
|
|
305
374
|
return results
|
|
306
375
|
|
|
307
376
|
|
|
308
|
-
def
|
|
377
|
+
def _content_value(note: dict, name: str):
|
|
378
|
+
value = (note.get("content") or {}).get(name)
|
|
379
|
+
return value.get("value") if isinstance(value, dict) and "value" in value else value
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _normalized_title(value: object) -> str:
|
|
383
|
+
return re.sub(r"\W+", "", unicodedata.normalize("NFKC", str(value or "")).casefold())
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _openreview_status(note: dict) -> str:
|
|
387
|
+
venue = _content_value(note, "venue") or _content_value(note, "venueid") or _content_value(note, "venue_id")
|
|
388
|
+
fields = [*note.get("invitations", []), venue, _content_value(note, "decision"), _content_value(note, "status")]
|
|
389
|
+
text = " ".join(str(value) for value in fields if value).casefold()
|
|
390
|
+
if "withdraw" in text:
|
|
391
|
+
return "withdrawn"
|
|
392
|
+
if "reject" in text:
|
|
393
|
+
return "rejected"
|
|
394
|
+
if "accept" in text or venue and not any(word in str(venue).casefold() for word in ("submission", "submitted")):
|
|
395
|
+
return "accepted"
|
|
396
|
+
return "submission"
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def search(
|
|
400
|
+
source: str,
|
|
401
|
+
query: str,
|
|
402
|
+
*,
|
|
403
|
+
limit: int = 10,
|
|
404
|
+
local_only: bool = False,
|
|
405
|
+
exact_title: bool = False,
|
|
406
|
+
openreview_status: str | None = None,
|
|
407
|
+
) -> dict:
|
|
408
|
+
query = query.strip()
|
|
409
|
+
if not query:
|
|
410
|
+
raise ValueError("paper search query is required")
|
|
411
|
+
if not 1 <= limit <= 100:
|
|
412
|
+
raise ValueError("limit must be between 1 and 100")
|
|
413
|
+
if (exact_title or openreview_status) and source != "openreview":
|
|
414
|
+
raise ValueError("exact title and OpenReview status filters require the openreview source")
|
|
415
|
+
if openreview_status not in (None, "submission", "accepted", "withdrawn"):
|
|
416
|
+
raise ValueError("OpenReview status must be submission, accepted, or withdrawn")
|
|
309
417
|
if source == "dblp":
|
|
310
418
|
from . import dblp_index
|
|
311
419
|
|
|
312
|
-
if hits := dblp_index.search(query):
|
|
420
|
+
if hits := dblp_index.search(query, limit=limit):
|
|
313
421
|
return _result("dblp", str(dblp_index.index_path()), {"query": query, "matches": hits})
|
|
314
422
|
if local_only:
|
|
315
423
|
return {
|
|
@@ -319,19 +427,25 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
|
|
|
319
427
|
"reason": "not found in local index",
|
|
320
428
|
}
|
|
321
429
|
url = "https://dblp.org/search/publ/api"
|
|
430
|
+
has_field_token = re.search(r"(?:^|\s)(?:author|title|venue|year|type|stream|toc):[^:]+:", query)
|
|
431
|
+
remote_query = query if has_field_token else re.sub(r":\s+", " ", query)
|
|
322
432
|
return _safe(
|
|
323
|
-
lambda: _result("dblp", url, _get_json(url, {"q":
|
|
433
|
+
lambda: _result("dblp", url, _get_json(url, {"q": remote_query, "format": "json", "h": limit})),
|
|
434
|
+
"dblp",
|
|
435
|
+
url,
|
|
324
436
|
)
|
|
325
437
|
if source == "crossref":
|
|
326
438
|
url = "https://api.crossref.org/works"
|
|
327
439
|
return _safe(
|
|
328
|
-
lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows":
|
|
440
|
+
lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows": limit})),
|
|
441
|
+
"crossref",
|
|
442
|
+
url,
|
|
329
443
|
)
|
|
330
444
|
if source == "arxiv":
|
|
331
445
|
url = "https://export.arxiv.org/api/query"
|
|
332
446
|
|
|
333
447
|
def arxiv_search() -> dict:
|
|
334
|
-
root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results":
|
|
448
|
+
root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results": limit}))
|
|
335
449
|
matches = [
|
|
336
450
|
{
|
|
337
451
|
"id": entry.findtext("atom:id", "", ARXIV_NS),
|
|
@@ -347,40 +461,75 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
|
|
|
347
461
|
token = os.environ.get("OPENREVIEW_ACCESS_TOKEN")
|
|
348
462
|
headers = {"Cookie": f"openreview.accessToken={token}"} if token else {}
|
|
349
463
|
endpoints = (
|
|
350
|
-
"https://api2.openreview.net/notes/search",
|
|
351
|
-
"https://api.openreview.net/notes/search",
|
|
464
|
+
("https://api2.openreview.net/notes/search", True),
|
|
465
|
+
("https://api.openreview.net/notes/search", False),
|
|
352
466
|
)
|
|
353
467
|
matches = []
|
|
354
468
|
errors = []
|
|
355
|
-
for endpoint in endpoints:
|
|
356
|
-
|
|
357
|
-
|
|
469
|
+
for endpoint, is_v2 in endpoints:
|
|
470
|
+
endpoint_matches = []
|
|
471
|
+
local_filter = bool(openreview_status or (exact_title and not is_v2))
|
|
472
|
+
page_size = min(max(limit * 2, 20), 100) if local_filter else limit
|
|
473
|
+
params = {"limit": page_size, "source": "forum", "cache": "true"}
|
|
474
|
+
if exact_title and is_v2:
|
|
475
|
+
params.update({"term": query, "type": "exact", "content": "title"})
|
|
476
|
+
else:
|
|
477
|
+
params["query"] = query
|
|
478
|
+
for offset in range(0, 1000, page_size):
|
|
479
|
+
page_params = {**params, "offset": offset} if offset else params
|
|
480
|
+
result = _safe(
|
|
481
|
+
lambda endpoint=endpoint, page_params=page_params: _result(
|
|
482
|
+
"openreview",
|
|
483
|
+
endpoint,
|
|
484
|
+
_get_json(endpoint, page_params, headers),
|
|
485
|
+
),
|
|
358
486
|
"openreview",
|
|
359
487
|
endpoint,
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
488
|
+
)
|
|
489
|
+
if result["status"] != "ok":
|
|
490
|
+
errors.append(result.get("error", endpoint))
|
|
491
|
+
break
|
|
492
|
+
notes = result.get("response", {}).get("notes", [])
|
|
493
|
+
matches.extend(notes)
|
|
494
|
+
endpoint_matches.extend(notes)
|
|
495
|
+
if not local_filter or len(notes) < page_size:
|
|
496
|
+
break
|
|
497
|
+
eligible = endpoint_matches
|
|
498
|
+
if exact_title:
|
|
499
|
+
wanted = _normalized_title(query)
|
|
500
|
+
eligible = [
|
|
501
|
+
note for note in eligible if _normalized_title(_content_value(note, "title")) == wanted
|
|
502
|
+
]
|
|
503
|
+
if openreview_status:
|
|
504
|
+
eligible = [note for note in eligible if _openreview_status(note) == openreview_status]
|
|
505
|
+
eligible_ids = {note.get("forum") or note.get("id") for note in eligible}
|
|
506
|
+
if len(eligible_ids) >= limit:
|
|
507
|
+
break
|
|
369
508
|
if not matches and errors:
|
|
370
|
-
return _result("openreview", endpoints[0], error="; ".join(errors))
|
|
509
|
+
return _result("openreview", endpoints[0][0], error="; ".join(errors))
|
|
371
510
|
unique = {}
|
|
372
511
|
for note in matches:
|
|
373
512
|
unique[note.get("forum") or note.get("id") or json.dumps(note, sort_keys=True)] = note
|
|
513
|
+
selected = list(unique.values())
|
|
514
|
+
if exact_title:
|
|
515
|
+
wanted = _normalized_title(query)
|
|
516
|
+
selected = [note for note in selected if _normalized_title(_content_value(note, "title")) == wanted]
|
|
517
|
+
if openreview_status:
|
|
518
|
+
selected = [note for note in selected if _openreview_status(note) == openreview_status]
|
|
374
519
|
return _result(
|
|
375
520
|
"openreview",
|
|
376
|
-
endpoints[0],
|
|
377
|
-
{
|
|
521
|
+
endpoints[0][0],
|
|
522
|
+
{
|
|
523
|
+
"query": query,
|
|
524
|
+
"matches": selected[:limit],
|
|
525
|
+
"api_endpoints": [endpoint for endpoint, _ in endpoints],
|
|
526
|
+
},
|
|
378
527
|
)
|
|
379
528
|
if source == "s2":
|
|
380
529
|
url = "https://api.semanticscholar.org/graph/v1/paper/search"
|
|
381
530
|
api_key = credentials.get(credentials.SEMANTIC_SCHOLAR_API_KEY)
|
|
382
531
|
headers = {"x-api-key": api_key} if api_key else {}
|
|
383
|
-
params = {"query": query, "limit":
|
|
532
|
+
params = {"query": query, "limit": limit, "fields": S2_FIELDS}
|
|
384
533
|
return _safe(
|
|
385
534
|
lambda: _result("semantic_scholar", url, _get_json(url, params, headers)), "semantic_scholar", url
|
|
386
535
|
)
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""TLS settings for hosts that reject some client handshakes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import functools
|
|
6
|
+
import ssl
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@functools.cache
|
|
10
|
+
def arxiv_context() -> ssl.SSLContext:
|
|
11
|
+
# arXiv's Fastly edge answers uncached requests with 406 when the TLS 1.3
|
|
12
|
+
# ClientHello offers the X25519MLKEM768 hybrid key share (OpenSSL >= 3.5 default).
|
|
13
|
+
context = ssl.create_default_context()
|
|
14
|
+
context.set_ecdh_curve("X25519")
|
|
15
|
+
return context
|
|
@@ -94,11 +94,12 @@ def build(root: Path, output: Path) -> int:
|
|
|
94
94
|
):
|
|
95
95
|
raise ValueError("viewer output would replace a broad path, the corpus, or authored entries")
|
|
96
96
|
backup = destination.parent / f".{destination.name}.backup"
|
|
97
|
-
if backup.exists()
|
|
97
|
+
if backup.exists():
|
|
98
98
|
marker = backup / ".paperstack-viewer"
|
|
99
99
|
if not backup.is_dir() or not marker.is_file() or marker.read_text(encoding="utf-8") != "1\n":
|
|
100
100
|
raise ValueError(f"viewer backup is not owned by Paperstack: {backup}")
|
|
101
|
-
|
|
101
|
+
if not destination.exists():
|
|
102
|
+
os.replace(backup, destination)
|
|
102
103
|
if destination.exists() and not destination.is_dir():
|
|
103
104
|
raise ValueError(f"viewer output is not a directory: {destination}")
|
|
104
105
|
if destination.exists() and any(destination.iterdir()):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{paperstack_cli-0.4.0 → paperstack_cli-0.4.2}/src/paperstack/content/vendor/latexpand.LICENSE
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|