paperstack-cli 0.4.1__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/PKG-INFO +16 -3
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/README.md +15 -2
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/cli.py +73 -38
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/arxiv_pdf.py +30 -8
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/arxiv_source.py +27 -28
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/metadata.py +8 -3
- paperstack_cli-0.5.0/src/paperstack/tls.py +15 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/.gitignore +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/LICENSE +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/pyproject.toml +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/app.js +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/entry.html +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/favicon.svg +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/index.html +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/style.css +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/vendor/marked.LICENSE +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/vendor/marked.min.js +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/__init__.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/arxiv.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/citations.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/__init__.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/vendor/latexpand +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/vendor/latexpand.LICENSE +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/corpora.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/credentials.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/dblp_build.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/dblp_catalog.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/dblp_index.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/entry_types.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/entrypoint.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/semantic_scholar.py +0 -0
- {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/viewer.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: paperstack-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Review, inspect, and retrieve research sources from one CLI
|
|
5
5
|
Project-URL: Repository, https://github.com/MilkClouds/paperstack
|
|
6
6
|
Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
|
|
@@ -146,12 +146,25 @@ paperstack paper read arxiv:2604.23073 --section 6
|
|
|
146
146
|
paperstack paper pdf arxiv:2602.09017
|
|
147
147
|
```
|
|
148
148
|
|
|
149
|
-
`metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
|
|
150
|
-
|
|
149
|
+
`metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
|
|
150
|
+
accept arXiv references directly; other identifiers require an explicit primary-source HTTPS `--pdf-url`. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
|
|
151
151
|
`pmid:`, and `mag:` identifiers. Structured commands expose scoped `--json` flags; networked commands expose scoped
|
|
152
152
|
`--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
|
|
153
153
|
of a work should be cited.
|
|
154
154
|
|
|
155
|
+
### Source reading
|
|
156
|
+
|
|
157
|
+
- Pin `arxiv:IDvN` for version-specific downloads and caches. Metadata lookup remains work-level.
|
|
158
|
+
Unversioned reads may use stale cached content; `--refresh` refetches, `--offline` never does.
|
|
159
|
+
- `read ... --documents` lists source files. Multiple TeX roots require `--document FILE`;
|
|
160
|
+
inspect relevant supplements separately. `--outline`/`--section` apply to that document.
|
|
161
|
+
- DOI/OpenReview/DBLP reads require `--pdf-url https://...` pointing to the primary PDF.
|
|
162
|
+
`read` prints Markdown; `pdf` prints its cache path. New conversions need the PDF extra.
|
|
163
|
+
- Partial PDF extraction returns failure unless `--allow-partial` is explicit, even offline.
|
|
164
|
+
Check `meta.json` and the PDF before using numbers. Converter quality does not certify accuracy.
|
|
165
|
+
- PDF caches verify original and Markdown hashes. Old caches without PDF hashes need rebuilding.
|
|
166
|
+
|
|
167
|
+
|
|
155
168
|
OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
|
|
156
169
|
filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
|
|
157
170
|
public invitation, venue, decision, and status fields rather than treating it as a universal field.
|
|
@@ -124,12 +124,25 @@ paperstack paper read arxiv:2604.23073 --section 6
|
|
|
124
124
|
paperstack paper pdf arxiv:2602.09017
|
|
125
125
|
```
|
|
126
126
|
|
|
127
|
-
`metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
|
|
128
|
-
|
|
127
|
+
`metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
|
|
128
|
+
accept arXiv references directly; other identifiers require an explicit primary-source HTTPS `--pdf-url`. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
|
|
129
129
|
`pmid:`, and `mag:` identifiers. Structured commands expose scoped `--json` flags; networked commands expose scoped
|
|
130
130
|
`--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
|
|
131
131
|
of a work should be cited.
|
|
132
132
|
|
|
133
|
+
### Source reading
|
|
134
|
+
|
|
135
|
+
- Pin `arxiv:IDvN` for version-specific downloads and caches. Metadata lookup remains work-level.
|
|
136
|
+
Unversioned reads may use stale cached content; `--refresh` refetches, `--offline` never does.
|
|
137
|
+
- `read ... --documents` lists source files. Multiple TeX roots require `--document FILE`;
|
|
138
|
+
inspect relevant supplements separately. `--outline`/`--section` apply to that document.
|
|
139
|
+
- DOI/OpenReview/DBLP reads require `--pdf-url https://...` pointing to the primary PDF.
|
|
140
|
+
`read` prints Markdown; `pdf` prints its cache path. New conversions need the PDF extra.
|
|
141
|
+
- Partial PDF extraction returns failure unless `--allow-partial` is explicit, even offline.
|
|
142
|
+
Check `meta.json` and the PDF before using numbers. Converter quality does not certify accuracy.
|
|
143
|
+
- PDF caches verify original and Markdown hashes. Old caches without PDF hashes need rebuilding.
|
|
144
|
+
|
|
145
|
+
|
|
133
146
|
OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
|
|
134
147
|
filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
|
|
135
148
|
public invitation, venue, decision, and status fields rather than treating it as a universal field.
|
|
@@ -872,6 +872,43 @@ def _paper_cache() -> Path:
|
|
|
872
872
|
return Path(os.environ.get("PAPERSTACK_PAPERS_DIR", base / "paperstack" / "papers"))
|
|
873
873
|
|
|
874
874
|
|
|
875
|
+
def _read_pdf(a, ref, offline):
|
|
876
|
+
import hashlib
|
|
877
|
+
from contextlib import redirect_stdout
|
|
878
|
+
from urllib.parse import urlparse
|
|
879
|
+
|
|
880
|
+
from .content import arxiv_pdf
|
|
881
|
+
from .content.arxiv_source import _print_chunk
|
|
882
|
+
|
|
883
|
+
url = a.pdf_url
|
|
884
|
+
if ref.kind != "arxiv" and not url:
|
|
885
|
+
die("non-arXiv content requires --pdf-url with the primary-source PDF URL")
|
|
886
|
+
if url and (urlparse(url).scheme != "https" or not urlparse(url).hostname or urlparse(url).username):
|
|
887
|
+
die("--pdf-url must be an HTTPS URL without credentials")
|
|
888
|
+
if offline and a.refresh:
|
|
889
|
+
die("--offline and --refresh cannot be used together")
|
|
890
|
+
if a.paper_cmd == "read" and (a.outline or a.section_id or a.documents or a.document):
|
|
891
|
+
die("PDF reads support --start/--max-chars, not LaTeX document or section selection")
|
|
892
|
+
key = ref.value if not url else "external-" + hashlib.sha256(f"{ref.kind}:{ref.value}\n{url}".encode()).hexdigest()
|
|
893
|
+
arxiv_pdf.CACHE_DIR = _paper_cache()
|
|
894
|
+
directory = arxiv_pdf.CACHE_DIR / key
|
|
895
|
+
if offline:
|
|
896
|
+
if arxiv_pdf._cached_conversion(directory, allow_partial=a.allow_partial) is None:
|
|
897
|
+
die("no usable cached PDF conversion; partial content requires --allow-partial")
|
|
898
|
+
else:
|
|
899
|
+
with redirect_stdout(sys.stderr):
|
|
900
|
+
converted = arxiv_pdf.convert(
|
|
901
|
+
key, url=url, refresh=a.refresh, allow_partial=a.allow_partial, paper_ref=f"{ref.kind}:{ref.value}"
|
|
902
|
+
)
|
|
903
|
+
if not converted:
|
|
904
|
+
return 1
|
|
905
|
+
if a.paper_cmd == "read":
|
|
906
|
+
_print_chunk((directory / "paper.md").read_text(), a.start, a.max_chars)
|
|
907
|
+
else:
|
|
908
|
+
print(directory / "paper.md")
|
|
909
|
+
return 0
|
|
910
|
+
|
|
911
|
+
|
|
875
912
|
def _run_paper(a: argparse.Namespace) -> int:
|
|
876
913
|
from . import credentials, metadata
|
|
877
914
|
|
|
@@ -1008,44 +1045,35 @@ def _run_paper(a: argparse.Namespace) -> int:
|
|
|
1008
1045
|
return 0 if result["status"] == "ok" else 1
|
|
1009
1046
|
|
|
1010
1047
|
try:
|
|
1011
|
-
ref = metadata.PaperRef.parse(a.paper_ref)
|
|
1048
|
+
ref = metadata.PaperRef.parse(a.paper_ref, preserve_version=True)
|
|
1012
1049
|
except ValueError as exc:
|
|
1013
1050
|
die(str(exc))
|
|
1014
|
-
if ref.kind
|
|
1015
|
-
|
|
1016
|
-
if a.paper_cmd == "
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
argv.
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
if a.
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
arxiv_pdf.CACHE_DIR / ref.value,
|
|
1041
|
-
allow_native_fallback=offline,
|
|
1042
|
-
)
|
|
1043
|
-
if offline and cached_pdf is None:
|
|
1044
|
-
die(f"no usable cached PDF conversion for arxiv:{ref.value}")
|
|
1045
|
-
if not arxiv_pdf.convert(ref.value, allow_native_fallback=offline):
|
|
1046
|
-
return 1
|
|
1047
|
-
return 0
|
|
1048
|
-
raise AssertionError(a.paper_cmd)
|
|
1051
|
+
if ref.kind == "arxiv" and not re.search(r"v[1-9]\d*$", ref.value):
|
|
1052
|
+
warn("Unversioned arXiv reference: cached content may not be current; pin vN for reproducible reads.")
|
|
1053
|
+
if a.pdf_url or a.paper_cmd == "pdf" or ref.kind != "arxiv":
|
|
1054
|
+
return _read_pdf(a, ref, offline)
|
|
1055
|
+
from .content import arxiv_source
|
|
1056
|
+
|
|
1057
|
+
arxiv_source.CACHE_DIR = _paper_cache()
|
|
1058
|
+
cached_source = arxiv_source.CACHE_DIR / ref.value / "src"
|
|
1059
|
+
if offline and a.refresh:
|
|
1060
|
+
die("--offline and --refresh cannot be used together")
|
|
1061
|
+
if offline and (not cached_source.is_dir() or not arxiv_source._tex_candidates(cached_source)):
|
|
1062
|
+
die(f"no complete cached source for arxiv:{ref.value}")
|
|
1063
|
+
argv = ["read", ref.value]
|
|
1064
|
+
if a.outline:
|
|
1065
|
+
argv.append("--outline")
|
|
1066
|
+
elif a.section_id:
|
|
1067
|
+
argv.extend(["--section", a.section_id])
|
|
1068
|
+
if a.documents:
|
|
1069
|
+
argv.append("--documents")
|
|
1070
|
+
if a.document:
|
|
1071
|
+
argv.extend(["--document", a.document])
|
|
1072
|
+
if a.refresh:
|
|
1073
|
+
argv.append("--refresh")
|
|
1074
|
+
argv.extend(["--start", str(a.start), "--max-chars", str(a.max_chars)])
|
|
1075
|
+
arxiv_source.main(argv)
|
|
1076
|
+
return 0
|
|
1049
1077
|
|
|
1050
1078
|
|
|
1051
1079
|
def _run_index(a: argparse.Namespace) -> int:
|
|
@@ -1226,16 +1254,23 @@ Use `paperstack review ...` to find or read an authored critical judgment.""",
|
|
|
1226
1254
|
_output(s)
|
|
1227
1255
|
_offline(s)
|
|
1228
1256
|
s = paper_sub.add_parser("read", help="read the LaTeX body, outline, or one section")
|
|
1229
|
-
s.add_argument("paper_ref", help="arxiv: reference")
|
|
1257
|
+
s.add_argument("paper_ref", help="arxiv: (version preserved), doi:, dblp:, or openreview: reference")
|
|
1230
1258
|
mode = s.add_mutually_exclusive_group()
|
|
1231
1259
|
mode.add_argument("--outline", action="store_true", help="print numbered section headings only")
|
|
1232
1260
|
mode.add_argument("--section", dest="section_id", help="print one section by outline number")
|
|
1233
1261
|
s.add_argument("--refresh", action="store_true", help="replace the cached arXiv source")
|
|
1234
1262
|
s.add_argument("--start", type=int, default=0, help="start at this character offset")
|
|
1235
1263
|
s.add_argument("--max-chars", type=int, default=0, help="truncate output after this many characters")
|
|
1264
|
+
s.add_argument("--documents", action="store_true", help="list source files, including supplements")
|
|
1265
|
+
s.add_argument("--document", help="source-relative TeX root to read")
|
|
1266
|
+
s.add_argument("--pdf-url", help="explicit primary-source HTTPS PDF URL")
|
|
1267
|
+
s.add_argument("--allow-partial", action="store_true", help="explicitly accept incomplete PDF extraction")
|
|
1236
1268
|
_offline(s)
|
|
1237
1269
|
s = paper_sub.add_parser("pdf", help="download and convert a native PDF submission")
|
|
1238
|
-
s.add_argument("paper_ref", help="arxiv: reference")
|
|
1270
|
+
s.add_argument("paper_ref", help="arxiv: (version preserved), doi:, dblp:, or openreview: reference")
|
|
1271
|
+
s.add_argument("--pdf-url", help="explicit primary-source HTTPS PDF URL")
|
|
1272
|
+
s.add_argument("--allow-partial", action="store_true", help="explicitly accept incomplete PDF extraction")
|
|
1273
|
+
s.add_argument("--refresh", action="store_true", help="download and convert the PDF again")
|
|
1239
1274
|
_offline(s)
|
|
1240
1275
|
|
|
1241
1276
|
index = sub.add_parser("index", help="optional local lookup indexes")
|
|
@@ -16,6 +16,8 @@ import urllib.error
|
|
|
16
16
|
import urllib.request
|
|
17
17
|
from pathlib import Path
|
|
18
18
|
|
|
19
|
+
from ..tls import arxiv_context
|
|
20
|
+
|
|
19
21
|
_CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
20
22
|
CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
|
|
21
23
|
CONVERTER = "pdf-inspector"
|
|
@@ -27,7 +29,7 @@ def _fetch(url: str, timeout: int = 60) -> bytes | None:
|
|
|
27
29
|
req = urllib.request.Request(url, headers={"User-Agent": "paperstack/1.0 (+arxiv pdf fetch)"})
|
|
28
30
|
for attempt in range(3):
|
|
29
31
|
try:
|
|
30
|
-
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
32
|
+
with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
|
|
31
33
|
return resp.read()
|
|
32
34
|
except urllib.error.HTTPError as e:
|
|
33
35
|
if e.code == 429:
|
|
@@ -43,7 +45,7 @@ def _fetch(url: str, timeout: int = 60) -> bytes | None:
|
|
|
43
45
|
return None
|
|
44
46
|
|
|
45
47
|
|
|
46
|
-
def _cached_conversion(d: Path, *,
|
|
48
|
+
def _cached_conversion(d: Path, *, allow_partial: bool = False) -> Path | None:
|
|
47
49
|
md_path = d / "paper.md"
|
|
48
50
|
meta_path = d / "meta.json"
|
|
49
51
|
if not md_path.is_file() or md_path.stat().st_size < MIN_MARKDOWN_CHARS or not meta_path.is_file():
|
|
@@ -59,7 +61,9 @@ def _cached_conversion(d: Path, *, allow_native_fallback: bool = False) -> Path
|
|
|
59
61
|
meta.get("converter") == CONVERTER
|
|
60
62
|
and meta.get("bytes") == len(markdown)
|
|
61
63
|
and meta.get("sha256") == hashlib.sha256(markdown).hexdigest()
|
|
62
|
-
and (
|
|
64
|
+
and (allow_partial or (meta.get("quality") == "complete" and meta.get("conversion_mode") != "native_fallback"))
|
|
65
|
+
and (d / "paper.pdf").is_file()
|
|
66
|
+
and meta.get("pdf_sha256") == hashlib.sha256((d / "paper.pdf").read_bytes()).hexdigest()
|
|
63
67
|
)
|
|
64
68
|
return md_path if valid else None
|
|
65
69
|
|
|
@@ -167,10 +171,17 @@ def _download_pdf(url: str, pdf_path: Path) -> bool:
|
|
|
167
171
|
return True
|
|
168
172
|
|
|
169
173
|
|
|
170
|
-
def convert(
|
|
174
|
+
def convert(
|
|
175
|
+
arxiv_id: str,
|
|
176
|
+
*,
|
|
177
|
+
allow_partial: bool = False,
|
|
178
|
+
url: str | None = None,
|
|
179
|
+
refresh: bool = False,
|
|
180
|
+
paper_ref: str | None = None,
|
|
181
|
+
) -> bool:
|
|
171
182
|
d = CACHE_DIR / arxiv_id
|
|
172
|
-
cached = _cached_conversion(d,
|
|
173
|
-
if cached is not None:
|
|
183
|
+
cached = _cached_conversion(d, allow_partial=allow_partial)
|
|
184
|
+
if cached is not None and not refresh:
|
|
174
185
|
print(f"{arxiv_id}: cached at {cached}")
|
|
175
186
|
return True
|
|
176
187
|
|
|
@@ -184,10 +195,10 @@ def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
|
|
|
184
195
|
)
|
|
185
196
|
return False
|
|
186
197
|
|
|
187
|
-
url = f"https://arxiv.org/pdf/{arxiv_id}"
|
|
198
|
+
url = url or f"https://arxiv.org/pdf/{arxiv_id}"
|
|
188
199
|
d.mkdir(parents=True, exist_ok=True)
|
|
189
200
|
pdf_path = d / "paper.pdf"
|
|
190
|
-
reused_cached_pdf = _is_pdf(pdf_path)
|
|
201
|
+
reused_cached_pdf = not refresh and _is_pdf(pdf_path)
|
|
191
202
|
if not reused_cached_pdf and not _download_pdf(url, pdf_path):
|
|
192
203
|
return False
|
|
193
204
|
|
|
@@ -233,10 +244,21 @@ def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
|
|
|
233
244
|
print(f"{arxiv_id}: converted to only {chars} chars, treat as a failure", file=sys.stderr)
|
|
234
245
|
return False
|
|
235
246
|
|
|
247
|
+
if paper_ref:
|
|
248
|
+
meta["paper_ref"] = paper_ref
|
|
249
|
+
if not paper_ref.startswith("arxiv:"):
|
|
250
|
+
meta.pop("arxiv_id", None)
|
|
251
|
+
meta["pdf_sha256"] = hashlib.sha256(pdf_path.read_bytes()).hexdigest()
|
|
236
252
|
md_path = d / "paper.md"
|
|
237
253
|
md_path.write_bytes(md.encode("utf-8"))
|
|
238
254
|
(d / "meta.json").write_text(json.dumps(meta, indent=2), encoding="utf-8")
|
|
239
255
|
print(f"{arxiv_id}: {len(md)} chars -> {md_path}")
|
|
256
|
+
if meta["quality"] != "complete" and not allow_partial:
|
|
257
|
+
print(
|
|
258
|
+
"Partial extraction retained; inspect meta.json and the PDF, or explicitly use --allow-partial.",
|
|
259
|
+
file=sys.stderr,
|
|
260
|
+
)
|
|
261
|
+
return False
|
|
240
262
|
return True
|
|
241
263
|
|
|
242
264
|
|
|
@@ -22,6 +22,8 @@ import urllib.request
|
|
|
22
22
|
import zlib
|
|
23
23
|
from pathlib import Path
|
|
24
24
|
|
|
25
|
+
from ..tls import arxiv_context
|
|
26
|
+
|
|
25
27
|
_CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
26
28
|
CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
|
|
27
29
|
|
|
@@ -43,7 +45,7 @@ def _fetch_bytes(url: str, timeout: int = 60) -> bytes | None:
|
|
|
43
45
|
)
|
|
44
46
|
for attempt in range(3):
|
|
45
47
|
try:
|
|
46
|
-
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
48
|
+
with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
|
|
47
49
|
return resp.read()
|
|
48
50
|
except urllib.error.HTTPError as e:
|
|
49
51
|
if e.code == 429:
|
|
@@ -323,34 +325,24 @@ def _clean_title(s: str, macros: dict[str, str] | None = None) -> str:
|
|
|
323
325
|
return re.sub(r"\s+", " ", s).strip()
|
|
324
326
|
|
|
325
327
|
|
|
326
|
-
def _load_document(arxiv_id: str, refresh: bool) -> tuple[str, list[dict]]:
|
|
327
|
-
src = _ensure_source(arxiv_id, refresh)
|
|
328
|
+
def _load_document(arxiv_id: str, refresh: bool, document: str | None = None) -> tuple[str, list[dict]]:
|
|
329
|
+
src = _ensure_source(arxiv_id, refresh).resolve()
|
|
328
330
|
cands = _tex_candidates(src)
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
if
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
flat = _flatten(p, src)
|
|
342
|
-
if flat is None:
|
|
343
|
-
continue
|
|
344
|
-
sections = _find_sections(flat)
|
|
345
|
-
scored.append((False, len(sections), -len(p.parts), flat, sections))
|
|
346
|
-
if not scored:
|
|
331
|
+
roots = [p for p in cands if r"\documentclass" in _mask_comments(_read(p)[:200_000])]
|
|
332
|
+
if document:
|
|
333
|
+
selected = (src / document).resolve()
|
|
334
|
+
if not selected.is_relative_to(src.resolve()) or selected not in [p.resolve() for p in cands]:
|
|
335
|
+
sys.exit(f"unknown source document: {document}")
|
|
336
|
+
cands = [selected]
|
|
337
|
+
elif len(roots or cands) > 1:
|
|
338
|
+
names = ", ".join(str(p.relative_to(src)) for p in (roots or cands))
|
|
339
|
+
sys.exit(f"multiple source documents: {names}; select --document (inspect each supplement separately)")
|
|
340
|
+
selected = cands[0] if document else (roots or cands)[0] if cands else None
|
|
341
|
+
flat = _flatten(selected, src) if selected else None
|
|
342
|
+
if flat is None:
|
|
347
343
|
sys.exit(f"{arxiv_id}: no usable .tex file in {src}")
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
sys.exit(f"{arxiv_id}: LaTeX source found but no \\section commands in it")
|
|
351
|
-
masked = _mask_comments(best[3])
|
|
352
|
-
lo, hi = _body_span(masked)
|
|
353
|
-
return best[3][lo:hi], best[4]
|
|
344
|
+
lo, hi = _body_span(_mask_comments(flat))
|
|
345
|
+
return flat[lo:hi], _find_sections(flat)
|
|
354
346
|
|
|
355
347
|
|
|
356
348
|
def _load(arxiv_id: str, refresh: bool) -> list[dict]:
|
|
@@ -387,7 +379,12 @@ def cmd_section(args: argparse.Namespace) -> None:
|
|
|
387
379
|
|
|
388
380
|
|
|
389
381
|
def cmd_read(args: argparse.Namespace) -> None:
|
|
390
|
-
|
|
382
|
+
if getattr(args, "documents", False):
|
|
383
|
+
src = _ensure_source(args.arxiv_id, args.refresh)
|
|
384
|
+
for path in _tex_candidates(src):
|
|
385
|
+
print(path.relative_to(src))
|
|
386
|
+
return
|
|
387
|
+
body, sections = _load_document(args.arxiv_id, args.refresh, getattr(args, "document", None))
|
|
391
388
|
if args.outline:
|
|
392
389
|
for section in sections:
|
|
393
390
|
print(f"{section['id']}\t{section['level']}\t{section['title']}")
|
|
@@ -426,6 +423,8 @@ def main(argv: list[str]) -> None:
|
|
|
426
423
|
mode.add_argument("--section", dest="section_id", help="dotted id (3.2) or exact title")
|
|
427
424
|
p_read.add_argument("--max-chars", type=int, default=0, help="0 = all remaining text")
|
|
428
425
|
p_read.add_argument("--start", type=int, default=0)
|
|
426
|
+
p_read.add_argument("--documents", action="store_true")
|
|
427
|
+
p_read.add_argument("--document")
|
|
429
428
|
p_read.add_argument("--refresh", action="store_true", help="refetch even if cached")
|
|
430
429
|
p_read.set_defaults(func=cmd_read)
|
|
431
430
|
|
|
@@ -20,6 +20,7 @@ from pathlib import Path
|
|
|
20
20
|
from filelock import FileLock
|
|
21
21
|
|
|
22
22
|
from . import credentials
|
|
23
|
+
from .tls import arxiv_context
|
|
23
24
|
|
|
24
25
|
ARXIV_NS = {"atom": "http://www.w3.org/2005/Atom", "arxiv": "http://arxiv.org/schemas/atom"}
|
|
25
26
|
SOURCES = ("semantic_scholar", "dblp", "crossref", "openreview", "acl_anthology", "arxiv")
|
|
@@ -36,18 +37,21 @@ class PaperRef:
|
|
|
36
37
|
value: str
|
|
37
38
|
|
|
38
39
|
@classmethod
|
|
39
|
-
def parse(cls, raw: str) -> PaperRef:
|
|
40
|
+
def parse(cls, raw: str, *, preserve_version: bool = False) -> PaperRef:
|
|
40
41
|
if ":" not in raw:
|
|
41
42
|
raise ValueError("paper reference needs a prefix: arxiv:, doi:, dblp:, or openreview:")
|
|
42
43
|
kind, value = raw.strip().split(":", 1)
|
|
43
44
|
if kind not in ("arxiv", "doi", "dblp", "openreview") or not value:
|
|
44
45
|
raise ValueError("paper reference needs a prefix: arxiv:, doi:, dblp:, or openreview:")
|
|
45
46
|
if kind == "arxiv":
|
|
46
|
-
|
|
47
|
+
versioned = value
|
|
48
|
+
value = re.sub(r"v[1-9]\d*$", "", value)
|
|
47
49
|
modern = re.fullmatch(r"\d{2}(?:0[1-9]|1[0-2])\.\d{4,5}", value)
|
|
48
50
|
legacy = re.fullmatch(r"[A-Za-z][A-Za-z.-]*/\d{2}(?:0[1-9]|1[0-2])\d{3}", value)
|
|
49
51
|
if not (modern or legacy):
|
|
50
52
|
raise ValueError("invalid arXiv reference")
|
|
53
|
+
if preserve_version:
|
|
54
|
+
value = versioned
|
|
51
55
|
return cls(kind, value)
|
|
52
56
|
|
|
53
57
|
|
|
@@ -123,6 +127,7 @@ def request(
|
|
|
123
127
|
**(headers or {}),
|
|
124
128
|
}
|
|
125
129
|
req = urllib.request.Request(url, headers=request_headers, data=data)
|
|
130
|
+
open_options = {"context": arxiv_context()} if "arxiv.org" in host else {}
|
|
126
131
|
with _request_slot(request_key) as shared_timestamp:
|
|
127
132
|
for attempt in range(3):
|
|
128
133
|
elapsed = min(
|
|
@@ -132,7 +137,7 @@ def request(
|
|
|
132
137
|
if elapsed < interval:
|
|
133
138
|
time.sleep(interval - elapsed)
|
|
134
139
|
try:
|
|
135
|
-
with urllib.request.urlopen(req, timeout=30) as response:
|
|
140
|
+
with urllib.request.urlopen(req, timeout=30, **open_options) as response:
|
|
136
141
|
_last_request[request_key] = time.monotonic()
|
|
137
142
|
_mark_request(shared_timestamp)
|
|
138
143
|
return response.read()
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""TLS settings for hosts that reject some client handshakes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import functools
|
|
6
|
+
import ssl
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@functools.cache
|
|
10
|
+
def arxiv_context() -> ssl.SSLContext:
|
|
11
|
+
# arXiv's Fastly edge answers uncached requests with 406 when the TLS 1.3
|
|
12
|
+
# ClientHello offers the X25519MLKEM768 hybrid key share (OpenSSL >= 3.5 default).
|
|
13
|
+
context = ssl.create_default_context()
|
|
14
|
+
context.set_ecdh_curve("X25519")
|
|
15
|
+
return context
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/vendor/latexpand.LICENSE
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|