paperstack-cli 0.4.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/PKG-INFO +6 -1
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/README.md +5 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/cli.py +55 -17
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/metadata.py +192 -45
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/viewer.py +3 -2
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/.gitignore +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/LICENSE +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/pyproject.toml +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/scripts/build/site/app.js +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/scripts/build/site/entry.html +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/scripts/build/site/favicon.svg +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/scripts/build/site/index.html +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/scripts/build/site/style.css +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/scripts/build/vendor/marked.LICENSE +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/scripts/build/vendor/marked.min.js +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/__init__.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/arxiv.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/citations.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/content/__init__.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/content/arxiv_pdf.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/content/arxiv_source.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/content/vendor/latexpand +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/content/vendor/latexpand.LICENSE +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/corpora.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/credentials.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/dblp_build.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/dblp_catalog.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/dblp_index.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/entry_types.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/entrypoint.py +0 -0
- {paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/semantic_scholar.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: paperstack-cli
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: Review, inspect, and retrieve research sources from one CLI
|
|
5
5
|
Project-URL: Repository, https://github.com/MilkClouds/paperstack
|
|
6
6
|
Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
|
|
@@ -134,6 +134,7 @@ These commands use external source records and do not select a citation or make
|
|
|
134
134
|
|
|
135
135
|
```bash
|
|
136
136
|
paperstack paper search "Attention Is All You Need" --source dblp
|
|
137
|
+
paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
|
|
137
138
|
paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
|
|
138
139
|
paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
|
|
139
140
|
paperstack paper metadata arxiv:2106.09685
|
|
@@ -151,6 +152,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
|
|
|
151
152
|
`--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
|
|
152
153
|
of a work should be cited.
|
|
153
154
|
|
|
155
|
+
OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
|
|
156
|
+
filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
|
|
157
|
+
public invitation, venue, decision, and status fields rather than treating it as a universal field.
|
|
158
|
+
|
|
154
159
|
## Build a viewer
|
|
155
160
|
|
|
156
161
|
```bash
|
|
@@ -112,6 +112,7 @@ These commands use external source records and do not select a citation or make
|
|
|
112
112
|
|
|
113
113
|
```bash
|
|
114
114
|
paperstack paper search "Attention Is All You Need" --source dblp
|
|
115
|
+
paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
|
|
115
116
|
paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
|
|
116
117
|
paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
|
|
117
118
|
paperstack paper metadata arxiv:2106.09685
|
|
@@ -129,6 +130,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
|
|
|
129
130
|
`--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
|
|
130
131
|
of a work should be cited.
|
|
131
132
|
|
|
133
|
+
OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
|
|
134
|
+
filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
|
|
135
|
+
public invitation, venue, decision, and status fields rather than treating it as a universal field.
|
|
136
|
+
|
|
132
137
|
## Build a viewer
|
|
133
138
|
|
|
134
139
|
```bash
|
|
@@ -925,12 +925,42 @@ def _run_paper(a: argparse.Namespace) -> int:
|
|
|
925
925
|
metadata.print_results(results, json_output=a.json)
|
|
926
926
|
return 0 if any(item["status"] == "ok" for item in results) else 1
|
|
927
927
|
if a.paper_cmd == "search":
|
|
928
|
+
semantic_filters = [
|
|
929
|
+
flag
|
|
930
|
+
for flag, selected in (
|
|
931
|
+
("--offset", a.offset),
|
|
932
|
+
("--year", a.year),
|
|
933
|
+
("--field-of-study", a.fields_of_study),
|
|
934
|
+
("--open-access", a.open_access),
|
|
935
|
+
)
|
|
936
|
+
if selected
|
|
937
|
+
]
|
|
938
|
+
arxiv_filters = [
|
|
939
|
+
flag
|
|
940
|
+
for flag, selected in (
|
|
941
|
+
("--category", a.categories),
|
|
942
|
+
("--date-from", a.date_from),
|
|
943
|
+
("--date-to", a.date_to),
|
|
944
|
+
("--sort", a.sort != "relevance"),
|
|
945
|
+
)
|
|
946
|
+
if selected
|
|
947
|
+
]
|
|
948
|
+
openreview_filters = [
|
|
949
|
+
flag
|
|
950
|
+
for flag, selected in (
|
|
951
|
+
("--exact-title", a.exact_title),
|
|
952
|
+
("--openreview-status", a.openreview_status),
|
|
953
|
+
)
|
|
954
|
+
if selected
|
|
955
|
+
]
|
|
928
956
|
try:
|
|
957
|
+
if a.source != "openreview" and openreview_filters:
|
|
958
|
+
die(f"--source openreview is required for {', '.join(openreview_filters)}")
|
|
929
959
|
if a.source == "arxiv":
|
|
930
960
|
from . import arxiv
|
|
931
961
|
|
|
932
|
-
if
|
|
933
|
-
die("--
|
|
962
|
+
if semantic_filters:
|
|
963
|
+
die(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
|
|
934
964
|
result = arxiv.search(
|
|
935
965
|
a.query,
|
|
936
966
|
categories=a.categories,
|
|
@@ -942,8 +972,8 @@ def _run_paper(a: argparse.Namespace) -> int:
|
|
|
942
972
|
elif a.source == "semantic-scholar":
|
|
943
973
|
from . import semantic_scholar
|
|
944
974
|
|
|
945
|
-
if
|
|
946
|
-
die("--
|
|
975
|
+
if arxiv_filters:
|
|
976
|
+
die(f"--source arxiv is required for {', '.join(arxiv_filters)}")
|
|
947
977
|
result = semantic_scholar.search(
|
|
948
978
|
a.query,
|
|
949
979
|
limit=a.limit,
|
|
@@ -953,19 +983,21 @@ def _run_paper(a: argparse.Namespace) -> int:
|
|
|
953
983
|
open_access=a.open_access,
|
|
954
984
|
)
|
|
955
985
|
else:
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
986
|
+
requirements = []
|
|
987
|
+
if semantic_filters:
|
|
988
|
+
requirements.append(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
|
|
989
|
+
if arxiv_filters:
|
|
990
|
+
requirements.append(f"--source arxiv is required for {', '.join(arxiv_filters)}")
|
|
991
|
+
if requirements:
|
|
992
|
+
die("; ".join(requirements))
|
|
993
|
+
result = metadata.search(
|
|
994
|
+
a.source,
|
|
995
|
+
a.query,
|
|
996
|
+
limit=a.limit,
|
|
997
|
+
local_only=offline,
|
|
998
|
+
exact_title=a.exact_title,
|
|
999
|
+
openreview_status=a.openreview_status,
|
|
1000
|
+
)
|
|
969
1001
|
except credentials.CredentialsError as exc:
|
|
970
1002
|
die(f"configuration failed: {exc}")
|
|
971
1003
|
except RuntimeError as exc:
|
|
@@ -1178,6 +1210,12 @@ Use `paperstack review ...` to find or read an authored critical judgment.""",
|
|
|
1178
1210
|
s.add_argument("--date-from", help="earliest arXiv submission date")
|
|
1179
1211
|
s.add_argument("--date-to", help="latest arXiv submission date")
|
|
1180
1212
|
s.add_argument("--sort", choices=("relevance", "date"), default="relevance")
|
|
1213
|
+
s.add_argument("--exact-title", action="store_true", help="require a normalized exact OpenReview title")
|
|
1214
|
+
s.add_argument(
|
|
1215
|
+
"--openreview-status",
|
|
1216
|
+
choices=("submission", "accepted", "withdrawn"),
|
|
1217
|
+
help="filter OpenReview forum records by inferred status",
|
|
1218
|
+
)
|
|
1181
1219
|
_output(s)
|
|
1182
1220
|
_offline(s)
|
|
1183
1221
|
for command in ("authors", "citations", "references"):
|
|
@@ -6,11 +6,18 @@ import json
|
|
|
6
6
|
import os
|
|
7
7
|
import re
|
|
8
8
|
import time
|
|
9
|
+
import unicodedata
|
|
9
10
|
import urllib.error
|
|
10
11
|
import urllib.parse
|
|
11
12
|
import urllib.request
|
|
12
13
|
import xml.etree.ElementTree as ET
|
|
14
|
+
from contextlib import contextmanager
|
|
13
15
|
from dataclasses import dataclass
|
|
16
|
+
from datetime import UTC, datetime
|
|
17
|
+
from email.utils import parsedate_to_datetime
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from filelock import FileLock
|
|
14
21
|
|
|
15
22
|
from . import credentials
|
|
16
23
|
|
|
@@ -45,6 +52,59 @@ class PaperRef:
|
|
|
45
52
|
|
|
46
53
|
|
|
47
54
|
_last_request: dict[str, float] = {}
|
|
55
|
+
_OPENREVIEW_HOSTS = {"api.openreview.net", "api2.openreview.net"}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@contextmanager
|
|
59
|
+
def _request_slot(key: str):
|
|
60
|
+
if key != "openreview":
|
|
61
|
+
yield None
|
|
62
|
+
return
|
|
63
|
+
base = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
64
|
+
root = base / "paperstack" / "http"
|
|
65
|
+
try:
|
|
66
|
+
root.mkdir(parents=True, exist_ok=True, mode=0o700)
|
|
67
|
+
lock = FileLock(root / "openreview.lock", timeout=120)
|
|
68
|
+
lock.acquire()
|
|
69
|
+
except OSError:
|
|
70
|
+
yield None
|
|
71
|
+
return
|
|
72
|
+
try:
|
|
73
|
+
yield root / "openreview.timestamp"
|
|
74
|
+
finally:
|
|
75
|
+
lock.release()
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _shared_elapsed(path: Path | None) -> float:
|
|
79
|
+
if path is None:
|
|
80
|
+
return float("inf")
|
|
81
|
+
try:
|
|
82
|
+
return max(0.0, time.time() - path.stat().st_mtime)
|
|
83
|
+
except OSError:
|
|
84
|
+
return float("inf")
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _mark_request(path: Path | None) -> None:
|
|
88
|
+
if path is not None:
|
|
89
|
+
try:
|
|
90
|
+
path.touch()
|
|
91
|
+
except OSError:
|
|
92
|
+
pass
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _retry_delay(exc: urllib.error.HTTPError, attempt: int) -> float:
|
|
96
|
+
value = exc.headers.get("Retry-After") if exc.headers else None
|
|
97
|
+
if value:
|
|
98
|
+
try:
|
|
99
|
+
delay = float(value)
|
|
100
|
+
except ValueError:
|
|
101
|
+
try:
|
|
102
|
+
delay = (parsedate_to_datetime(value) - datetime.now(UTC)).total_seconds()
|
|
103
|
+
except (TypeError, ValueError, OverflowError):
|
|
104
|
+
delay = -1
|
|
105
|
+
if delay >= 0:
|
|
106
|
+
return min(delay, 120)
|
|
107
|
+
return min(2**attempt, 30)
|
|
48
108
|
|
|
49
109
|
|
|
50
110
|
def request(
|
|
@@ -56,34 +116,41 @@ def request(
|
|
|
56
116
|
if params:
|
|
57
117
|
url += "?" + urllib.parse.urlencode(params)
|
|
58
118
|
host = urllib.parse.urlparse(url).netloc
|
|
119
|
+
request_key = "openreview" if host in _OPENREVIEW_HOSTS else host
|
|
59
120
|
interval = 3.0 if "arxiv.org" in host else 1.1 if "dblp.org" in host else 0.5
|
|
60
121
|
request_headers = {
|
|
61
122
|
"User-Agent": "paperstack (+https://github.com/MilkClouds/paperstack)",
|
|
62
123
|
**(headers or {}),
|
|
63
124
|
}
|
|
64
125
|
req = urllib.request.Request(url, headers=request_headers, data=data)
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
126
|
+
with _request_slot(request_key) as shared_timestamp:
|
|
127
|
+
for attempt in range(3):
|
|
128
|
+
elapsed = min(
|
|
129
|
+
time.monotonic() - _last_request.get(request_key, 0.0),
|
|
130
|
+
_shared_elapsed(shared_timestamp),
|
|
131
|
+
)
|
|
132
|
+
if elapsed < interval:
|
|
133
|
+
time.sleep(interval - elapsed)
|
|
134
|
+
try:
|
|
135
|
+
with urllib.request.urlopen(req, timeout=30) as response:
|
|
136
|
+
_last_request[request_key] = time.monotonic()
|
|
137
|
+
_mark_request(shared_timestamp)
|
|
138
|
+
return response.read()
|
|
139
|
+
except urllib.error.HTTPError as exc:
|
|
140
|
+
_last_request[request_key] = time.monotonic()
|
|
141
|
+
_mark_request(shared_timestamp)
|
|
142
|
+
if exc.code != 429:
|
|
143
|
+
raise
|
|
144
|
+
if attempt == 2:
|
|
145
|
+
has_api_key = any(name.lower() == "x-api-key" for name in request_headers)
|
|
146
|
+
if host == "api.semanticscholar.org" and not has_api_key:
|
|
147
|
+
message = (
|
|
148
|
+
f"{exc.reason}; configure semantic-scholar.api-key with "
|
|
149
|
+
"`paperstack config set semantic-scholar.api-key` for more reliable access"
|
|
150
|
+
)
|
|
151
|
+
raise urllib.error.HTTPError(exc.url, exc.code, message, exc.headers, exc.fp) from exc
|
|
152
|
+
raise
|
|
153
|
+
time.sleep(_retry_delay(exc, attempt))
|
|
87
154
|
raise RuntimeError("unreachable request retry state")
|
|
88
155
|
|
|
89
156
|
|
|
@@ -305,11 +372,50 @@ def fetch_all(
|
|
|
305
372
|
return results
|
|
306
373
|
|
|
307
374
|
|
|
308
|
-
def
|
|
375
|
+
def _content_value(note: dict, name: str):
|
|
376
|
+
value = (note.get("content") or {}).get(name)
|
|
377
|
+
return value.get("value") if isinstance(value, dict) and "value" in value else value
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _normalized_title(value: object) -> str:
|
|
381
|
+
return re.sub(r"\W+", "", unicodedata.normalize("NFKC", str(value or "")).casefold())
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _openreview_status(note: dict) -> str:
|
|
385
|
+
venue = _content_value(note, "venue") or _content_value(note, "venueid") or _content_value(note, "venue_id")
|
|
386
|
+
fields = [*note.get("invitations", []), venue, _content_value(note, "decision"), _content_value(note, "status")]
|
|
387
|
+
text = " ".join(str(value) for value in fields if value).casefold()
|
|
388
|
+
if "withdraw" in text:
|
|
389
|
+
return "withdrawn"
|
|
390
|
+
if "reject" in text:
|
|
391
|
+
return "rejected"
|
|
392
|
+
if "accept" in text or venue and not any(word in str(venue).casefold() for word in ("submission", "submitted")):
|
|
393
|
+
return "accepted"
|
|
394
|
+
return "submission"
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def search(
|
|
398
|
+
source: str,
|
|
399
|
+
query: str,
|
|
400
|
+
*,
|
|
401
|
+
limit: int = 10,
|
|
402
|
+
local_only: bool = False,
|
|
403
|
+
exact_title: bool = False,
|
|
404
|
+
openreview_status: str | None = None,
|
|
405
|
+
) -> dict:
|
|
406
|
+
query = query.strip()
|
|
407
|
+
if not query:
|
|
408
|
+
raise ValueError("paper search query is required")
|
|
409
|
+
if not 1 <= limit <= 100:
|
|
410
|
+
raise ValueError("limit must be between 1 and 100")
|
|
411
|
+
if (exact_title or openreview_status) and source != "openreview":
|
|
412
|
+
raise ValueError("exact title and OpenReview status filters require the openreview source")
|
|
413
|
+
if openreview_status not in (None, "submission", "accepted", "withdrawn"):
|
|
414
|
+
raise ValueError("OpenReview status must be submission, accepted, or withdrawn")
|
|
309
415
|
if source == "dblp":
|
|
310
416
|
from . import dblp_index
|
|
311
417
|
|
|
312
|
-
if hits := dblp_index.search(query):
|
|
418
|
+
if hits := dblp_index.search(query, limit=limit):
|
|
313
419
|
return _result("dblp", str(dblp_index.index_path()), {"query": query, "matches": hits})
|
|
314
420
|
if local_only:
|
|
315
421
|
return {
|
|
@@ -319,19 +425,25 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
|
|
|
319
425
|
"reason": "not found in local index",
|
|
320
426
|
}
|
|
321
427
|
url = "https://dblp.org/search/publ/api"
|
|
428
|
+
has_field_token = re.search(r"(?:^|\s)(?:author|title|venue|year|type|stream|toc):[^:]+:", query)
|
|
429
|
+
remote_query = query if has_field_token else re.sub(r":\s+", " ", query)
|
|
322
430
|
return _safe(
|
|
323
|
-
lambda: _result("dblp", url, _get_json(url, {"q":
|
|
431
|
+
lambda: _result("dblp", url, _get_json(url, {"q": remote_query, "format": "json", "h": limit})),
|
|
432
|
+
"dblp",
|
|
433
|
+
url,
|
|
324
434
|
)
|
|
325
435
|
if source == "crossref":
|
|
326
436
|
url = "https://api.crossref.org/works"
|
|
327
437
|
return _safe(
|
|
328
|
-
lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows":
|
|
438
|
+
lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows": limit})),
|
|
439
|
+
"crossref",
|
|
440
|
+
url,
|
|
329
441
|
)
|
|
330
442
|
if source == "arxiv":
|
|
331
443
|
url = "https://export.arxiv.org/api/query"
|
|
332
444
|
|
|
333
445
|
def arxiv_search() -> dict:
|
|
334
|
-
root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results":
|
|
446
|
+
root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results": limit}))
|
|
335
447
|
matches = [
|
|
336
448
|
{
|
|
337
449
|
"id": entry.findtext("atom:id", "", ARXIV_NS),
|
|
@@ -347,40 +459,75 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
|
|
|
347
459
|
token = os.environ.get("OPENREVIEW_ACCESS_TOKEN")
|
|
348
460
|
headers = {"Cookie": f"openreview.accessToken={token}"} if token else {}
|
|
349
461
|
endpoints = (
|
|
350
|
-
"https://api2.openreview.net/notes/search",
|
|
351
|
-
"https://api.openreview.net/notes/search",
|
|
462
|
+
("https://api2.openreview.net/notes/search", True),
|
|
463
|
+
("https://api.openreview.net/notes/search", False),
|
|
352
464
|
)
|
|
353
465
|
matches = []
|
|
354
466
|
errors = []
|
|
355
|
-
for endpoint in endpoints:
|
|
356
|
-
|
|
357
|
-
|
|
467
|
+
for endpoint, is_v2 in endpoints:
|
|
468
|
+
endpoint_matches = []
|
|
469
|
+
local_filter = bool(openreview_status or (exact_title and not is_v2))
|
|
470
|
+
page_size = min(max(limit * 2, 20), 100) if local_filter else limit
|
|
471
|
+
params = {"limit": page_size, "source": "forum", "cache": "true"}
|
|
472
|
+
if exact_title and is_v2:
|
|
473
|
+
params.update({"term": query, "type": "exact", "content": "title"})
|
|
474
|
+
else:
|
|
475
|
+
params["query"] = query
|
|
476
|
+
for offset in range(0, 1000, page_size):
|
|
477
|
+
page_params = {**params, "offset": offset} if offset else params
|
|
478
|
+
result = _safe(
|
|
479
|
+
lambda endpoint=endpoint, page_params=page_params: _result(
|
|
480
|
+
"openreview",
|
|
481
|
+
endpoint,
|
|
482
|
+
_get_json(endpoint, page_params, headers),
|
|
483
|
+
),
|
|
358
484
|
"openreview",
|
|
359
485
|
endpoint,
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
486
|
+
)
|
|
487
|
+
if result["status"] != "ok":
|
|
488
|
+
errors.append(result.get("error", endpoint))
|
|
489
|
+
break
|
|
490
|
+
notes = result.get("response", {}).get("notes", [])
|
|
491
|
+
matches.extend(notes)
|
|
492
|
+
endpoint_matches.extend(notes)
|
|
493
|
+
if not local_filter or len(notes) < page_size:
|
|
494
|
+
break
|
|
495
|
+
eligible = endpoint_matches
|
|
496
|
+
if exact_title:
|
|
497
|
+
wanted = _normalized_title(query)
|
|
498
|
+
eligible = [
|
|
499
|
+
note for note in eligible if _normalized_title(_content_value(note, "title")) == wanted
|
|
500
|
+
]
|
|
501
|
+
if openreview_status:
|
|
502
|
+
eligible = [note for note in eligible if _openreview_status(note) == openreview_status]
|
|
503
|
+
eligible_ids = {note.get("forum") or note.get("id") for note in eligible}
|
|
504
|
+
if len(eligible_ids) >= limit:
|
|
505
|
+
break
|
|
369
506
|
if not matches and errors:
|
|
370
|
-
return _result("openreview", endpoints[0], error="; ".join(errors))
|
|
507
|
+
return _result("openreview", endpoints[0][0], error="; ".join(errors))
|
|
371
508
|
unique = {}
|
|
372
509
|
for note in matches:
|
|
373
510
|
unique[note.get("forum") or note.get("id") or json.dumps(note, sort_keys=True)] = note
|
|
511
|
+
selected = list(unique.values())
|
|
512
|
+
if exact_title:
|
|
513
|
+
wanted = _normalized_title(query)
|
|
514
|
+
selected = [note for note in selected if _normalized_title(_content_value(note, "title")) == wanted]
|
|
515
|
+
if openreview_status:
|
|
516
|
+
selected = [note for note in selected if _openreview_status(note) == openreview_status]
|
|
374
517
|
return _result(
|
|
375
518
|
"openreview",
|
|
376
|
-
endpoints[0],
|
|
377
|
-
{
|
|
519
|
+
endpoints[0][0],
|
|
520
|
+
{
|
|
521
|
+
"query": query,
|
|
522
|
+
"matches": selected[:limit],
|
|
523
|
+
"api_endpoints": [endpoint for endpoint, _ in endpoints],
|
|
524
|
+
},
|
|
378
525
|
)
|
|
379
526
|
if source == "s2":
|
|
380
527
|
url = "https://api.semanticscholar.org/graph/v1/paper/search"
|
|
381
528
|
api_key = credentials.get(credentials.SEMANTIC_SCHOLAR_API_KEY)
|
|
382
529
|
headers = {"x-api-key": api_key} if api_key else {}
|
|
383
|
-
params = {"query": query, "limit":
|
|
530
|
+
params = {"query": query, "limit": limit, "fields": S2_FIELDS}
|
|
384
531
|
return _safe(
|
|
385
532
|
lambda: _result("semantic_scholar", url, _get_json(url, params, headers)), "semantic_scholar", url
|
|
386
533
|
)
|
|
@@ -94,11 +94,12 @@ def build(root: Path, output: Path) -> int:
|
|
|
94
94
|
):
|
|
95
95
|
raise ValueError("viewer output would replace a broad path, the corpus, or authored entries")
|
|
96
96
|
backup = destination.parent / f".{destination.name}.backup"
|
|
97
|
-
if backup.exists()
|
|
97
|
+
if backup.exists():
|
|
98
98
|
marker = backup / ".paperstack-viewer"
|
|
99
99
|
if not backup.is_dir() or not marker.is_file() or marker.read_text(encoding="utf-8") != "1\n":
|
|
100
100
|
raise ValueError(f"viewer backup is not owned by Paperstack: {backup}")
|
|
101
|
-
|
|
101
|
+
if not destination.exists():
|
|
102
|
+
os.replace(backup, destination)
|
|
102
103
|
if destination.exists() and not destination.is_dir():
|
|
103
104
|
raise ValueError(f"viewer output is not a directory: {destination}")
|
|
104
105
|
if destination.exists() and any(destination.iterdir()):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{paperstack_cli-0.4.0 → paperstack_cli-0.4.1}/src/paperstack/content/vendor/latexpand.LICENSE
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|