paperstack-cli 0.4.1__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/PKG-INFO +16 -3
  2. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/README.md +15 -2
  3. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/cli.py +73 -38
  4. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/arxiv_pdf.py +30 -8
  5. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/arxiv_source.py +27 -28
  6. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/metadata.py +8 -3
  7. paperstack_cli-0.5.0/src/paperstack/tls.py +15 -0
  8. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/.gitignore +0 -0
  9. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/LICENSE +0 -0
  10. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/pyproject.toml +0 -0
  11. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/app.js +0 -0
  12. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/entry.html +0 -0
  13. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/favicon.svg +0 -0
  14. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/index.html +0 -0
  15. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/site/style.css +0 -0
  16. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/vendor/marked.LICENSE +0 -0
  17. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/scripts/build/vendor/marked.min.js +0 -0
  18. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/__init__.py +0 -0
  19. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/arxiv.py +0 -0
  20. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/citations.py +0 -0
  21. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/__init__.py +0 -0
  22. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/vendor/latexpand +0 -0
  23. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/content/vendor/latexpand.LICENSE +0 -0
  24. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/corpora.py +0 -0
  25. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/credentials.py +0 -0
  26. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/dblp_build.py +0 -0
  27. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/dblp_catalog.py +0 -0
  28. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/dblp_index.py +0 -0
  29. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/entry_types.py +0 -0
  30. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/entrypoint.py +0 -0
  31. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/semantic_scholar.py +0 -0
  32. {paperstack_cli-0.4.1 → paperstack_cli-0.5.0}/src/paperstack/viewer.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: paperstack-cli
3
- Version: 0.4.1
3
+ Version: 0.5.0
4
4
  Summary: Review, inspect, and retrieve research sources from one CLI
5
5
  Project-URL: Repository, https://github.com/MilkClouds/paperstack
6
6
  Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
@@ -146,12 +146,25 @@ paperstack paper read arxiv:2604.23073 --section 6
146
146
  paperstack paper pdf arxiv:2602.09017
147
147
  ```
148
148
 
149
- `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf` require an `arxiv:`
150
- reference. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
149
+ `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
150
+ accept arXiv references directly; other identifiers require an explicit primary-source HTTPS `--pdf-url`. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
151
151
  `pmid:`, and `mag:` identifiers. Structured commands expose scoped `--json` flags; networked commands expose scoped
152
152
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
153
153
  of a work should be cited.
154
154
 
155
+ ### Source reading
156
+
157
+ - Pin `arxiv:IDvN` for version-specific downloads and caches. Metadata lookup remains work-level.
158
+ Unversioned reads may use stale cached content; `--refresh` refetches, `--offline` never does.
159
+ - `read ... --documents` lists source files. Multiple TeX roots require `--document FILE`;
160
+ inspect relevant supplements separately. `--outline`/`--section` apply to that document.
161
+ - DOI/OpenReview/DBLP reads require `--pdf-url https://...` pointing to the primary PDF.
162
+ `read` prints Markdown; `pdf` prints its cache path. New conversions need the PDF extra.
163
+ - Partial PDF extraction returns failure unless `--allow-partial` is explicit, even offline.
164
+ Check `meta.json` and the PDF before using numbers. Converter quality does not certify accuracy.
165
+ - PDF caches verify original and Markdown hashes. Old caches without PDF hashes need rebuilding.
166
+
167
+
155
168
  OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
156
169
  filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
157
170
  public invitation, venue, decision, and status fields rather than treating it as a universal field.
@@ -124,12 +124,25 @@ paperstack paper read arxiv:2604.23073 --section 6
124
124
  paperstack paper pdf arxiv:2602.09017
125
125
  ```
126
126
 
127
- `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf` require an `arxiv:`
128
- reference. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
127
+ `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
128
+ accept arXiv references directly; other identifiers require an explicit primary-source HTTPS `--pdf-url`. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
129
129
  `pmid:`, and `mag:` identifiers. Structured commands expose scoped `--json` flags; networked commands expose scoped
130
130
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
131
131
  of a work should be cited.
132
132
 
133
+ ### Source reading
134
+
135
+ - Pin `arxiv:IDvN` for version-specific downloads and caches. Metadata lookup remains work-level.
136
+ Unversioned reads may use stale cached content; `--refresh` refetches, `--offline` never does.
137
+ - `read ... --documents` lists source files. Multiple TeX roots require `--document FILE`;
138
+ inspect relevant supplements separately. `--outline`/`--section` apply to that document.
139
+ - DOI/OpenReview/DBLP reads require `--pdf-url https://...` pointing to the primary PDF.
140
+ `read` prints Markdown; `pdf` prints its cache path. New conversions need the PDF extra.
141
+ - Partial PDF extraction returns failure unless `--allow-partial` is explicit, even offline.
142
+ Check `meta.json` and the PDF before using numbers. Converter quality does not certify accuracy.
143
+ - PDF caches verify original and Markdown hashes. Old caches without PDF hashes need rebuilding.
144
+
145
+
133
146
  OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
134
147
  filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
135
148
  public invitation, venue, decision, and status fields rather than treating it as a universal field.
@@ -872,6 +872,43 @@ def _paper_cache() -> Path:
872
872
  return Path(os.environ.get("PAPERSTACK_PAPERS_DIR", base / "paperstack" / "papers"))
873
873
 
874
874
 
875
+ def _read_pdf(a, ref, offline):
876
+ import hashlib
877
+ from contextlib import redirect_stdout
878
+ from urllib.parse import urlparse
879
+
880
+ from .content import arxiv_pdf
881
+ from .content.arxiv_source import _print_chunk
882
+
883
+ url = a.pdf_url
884
+ if ref.kind != "arxiv" and not url:
885
+ die("non-arXiv content requires --pdf-url with the primary-source PDF URL")
886
+ if url and (urlparse(url).scheme != "https" or not urlparse(url).hostname or urlparse(url).username):
887
+ die("--pdf-url must be an HTTPS URL without credentials")
888
+ if offline and a.refresh:
889
+ die("--offline and --refresh cannot be used together")
890
+ if a.paper_cmd == "read" and (a.outline or a.section_id or a.documents or a.document):
891
+ die("PDF reads support --start/--max-chars, not LaTeX document or section selection")
892
+ key = ref.value if not url else "external-" + hashlib.sha256(f"{ref.kind}:{ref.value}\n{url}".encode()).hexdigest()
893
+ arxiv_pdf.CACHE_DIR = _paper_cache()
894
+ directory = arxiv_pdf.CACHE_DIR / key
895
+ if offline:
896
+ if arxiv_pdf._cached_conversion(directory, allow_partial=a.allow_partial) is None:
897
+ die("no usable cached PDF conversion; partial content requires --allow-partial")
898
+ else:
899
+ with redirect_stdout(sys.stderr):
900
+ converted = arxiv_pdf.convert(
901
+ key, url=url, refresh=a.refresh, allow_partial=a.allow_partial, paper_ref=f"{ref.kind}:{ref.value}"
902
+ )
903
+ if not converted:
904
+ return 1
905
+ if a.paper_cmd == "read":
906
+ _print_chunk((directory / "paper.md").read_text(), a.start, a.max_chars)
907
+ else:
908
+ print(directory / "paper.md")
909
+ return 0
910
+
911
+
875
912
  def _run_paper(a: argparse.Namespace) -> int:
876
913
  from . import credentials, metadata
877
914
 
@@ -1008,44 +1045,35 @@ def _run_paper(a: argparse.Namespace) -> int:
1008
1045
  return 0 if result["status"] == "ok" else 1
1009
1046
 
1010
1047
  try:
1011
- ref = metadata.PaperRef.parse(a.paper_ref)
1048
+ ref = metadata.PaperRef.parse(a.paper_ref, preserve_version=True)
1012
1049
  except ValueError as exc:
1013
1050
  die(str(exc))
1014
- if ref.kind != "arxiv":
1015
- die(f"paper {a.paper_cmd} currently requires an arxiv: reference")
1016
- if a.paper_cmd == "read":
1017
- from .content import arxiv_source
1018
-
1019
- arxiv_source.CACHE_DIR = _paper_cache()
1020
- cached_source = arxiv_source.CACHE_DIR / ref.value / "src"
1021
- if offline and a.refresh:
1022
- die("--offline and --refresh cannot be used together")
1023
- if offline and (not cached_source.is_dir() or not arxiv_source._tex_candidates(cached_source)):
1024
- die(f"no complete cached source for arxiv:{ref.value}")
1025
- argv = ["read", ref.value]
1026
- if a.outline:
1027
- argv.append("--outline")
1028
- elif a.section_id:
1029
- argv.extend(["--section", a.section_id])
1030
- if a.refresh:
1031
- argv.append("--refresh")
1032
- argv.extend(["--start", str(a.start), "--max-chars", str(a.max_chars)])
1033
- arxiv_source.main(argv)
1034
- return 0
1035
- if a.paper_cmd == "pdf":
1036
- from .content import arxiv_pdf
1037
-
1038
- arxiv_pdf.CACHE_DIR = _paper_cache()
1039
- cached_pdf = arxiv_pdf._cached_conversion(
1040
- arxiv_pdf.CACHE_DIR / ref.value,
1041
- allow_native_fallback=offline,
1042
- )
1043
- if offline and cached_pdf is None:
1044
- die(f"no usable cached PDF conversion for arxiv:{ref.value}")
1045
- if not arxiv_pdf.convert(ref.value, allow_native_fallback=offline):
1046
- return 1
1047
- return 0
1048
- raise AssertionError(a.paper_cmd)
1051
+ if ref.kind == "arxiv" and not re.search(r"v[1-9]\d*$", ref.value):
1052
+ warn("Unversioned arXiv reference: cached content may not be current; pin vN for reproducible reads.")
1053
+ if a.pdf_url or a.paper_cmd == "pdf" or ref.kind != "arxiv":
1054
+ return _read_pdf(a, ref, offline)
1055
+ from .content import arxiv_source
1056
+
1057
+ arxiv_source.CACHE_DIR = _paper_cache()
1058
+ cached_source = arxiv_source.CACHE_DIR / ref.value / "src"
1059
+ if offline and a.refresh:
1060
+ die("--offline and --refresh cannot be used together")
1061
+ if offline and (not cached_source.is_dir() or not arxiv_source._tex_candidates(cached_source)):
1062
+ die(f"no complete cached source for arxiv:{ref.value}")
1063
+ argv = ["read", ref.value]
1064
+ if a.outline:
1065
+ argv.append("--outline")
1066
+ elif a.section_id:
1067
+ argv.extend(["--section", a.section_id])
1068
+ if a.documents:
1069
+ argv.append("--documents")
1070
+ if a.document:
1071
+ argv.extend(["--document", a.document])
1072
+ if a.refresh:
1073
+ argv.append("--refresh")
1074
+ argv.extend(["--start", str(a.start), "--max-chars", str(a.max_chars)])
1075
+ arxiv_source.main(argv)
1076
+ return 0
1049
1077
 
1050
1078
 
1051
1079
  def _run_index(a: argparse.Namespace) -> int:
@@ -1226,16 +1254,23 @@ Use `paperstack review ...` to find or read an authored critical judgment.""",
1226
1254
  _output(s)
1227
1255
  _offline(s)
1228
1256
  s = paper_sub.add_parser("read", help="read the LaTeX body, outline, or one section")
1229
- s.add_argument("paper_ref", help="arxiv: reference")
1257
+ s.add_argument("paper_ref", help="arxiv: (version preserved), doi:, dblp:, or openreview: reference")
1230
1258
  mode = s.add_mutually_exclusive_group()
1231
1259
  mode.add_argument("--outline", action="store_true", help="print numbered section headings only")
1232
1260
  mode.add_argument("--section", dest="section_id", help="print one section by outline number")
1233
1261
  s.add_argument("--refresh", action="store_true", help="replace the cached arXiv source")
1234
1262
  s.add_argument("--start", type=int, default=0, help="start at this character offset")
1235
1263
  s.add_argument("--max-chars", type=int, default=0, help="truncate output after this many characters")
1264
+ s.add_argument("--documents", action="store_true", help="list source files, including supplements")
1265
+ s.add_argument("--document", help="source-relative TeX root to read")
1266
+ s.add_argument("--pdf-url", help="explicit primary-source HTTPS PDF URL")
1267
+ s.add_argument("--allow-partial", action="store_true", help="explicitly accept incomplete PDF extraction")
1236
1268
  _offline(s)
1237
1269
  s = paper_sub.add_parser("pdf", help="download and convert a native PDF submission")
1238
- s.add_argument("paper_ref", help="arxiv: reference")
1270
+ s.add_argument("paper_ref", help="arxiv: (version preserved), doi:, dblp:, or openreview: reference")
1271
+ s.add_argument("--pdf-url", help="explicit primary-source HTTPS PDF URL")
1272
+ s.add_argument("--allow-partial", action="store_true", help="explicitly accept incomplete PDF extraction")
1273
+ s.add_argument("--refresh", action="store_true", help="download and convert the PDF again")
1239
1274
  _offline(s)
1240
1275
 
1241
1276
  index = sub.add_parser("index", help="optional local lookup indexes")
@@ -16,6 +16,8 @@ import urllib.error
16
16
  import urllib.request
17
17
  from pathlib import Path
18
18
 
19
+ from ..tls import arxiv_context
20
+
19
21
  _CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
20
22
  CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
21
23
  CONVERTER = "pdf-inspector"
@@ -27,7 +29,7 @@ def _fetch(url: str, timeout: int = 60) -> bytes | None:
27
29
  req = urllib.request.Request(url, headers={"User-Agent": "paperstack/1.0 (+arxiv pdf fetch)"})
28
30
  for attempt in range(3):
29
31
  try:
30
- with urllib.request.urlopen(req, timeout=timeout) as resp:
32
+ with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
31
33
  return resp.read()
32
34
  except urllib.error.HTTPError as e:
33
35
  if e.code == 429:
@@ -43,7 +45,7 @@ def _fetch(url: str, timeout: int = 60) -> bytes | None:
43
45
  return None
44
46
 
45
47
 
46
- def _cached_conversion(d: Path, *, allow_native_fallback: bool = False) -> Path | None:
48
+ def _cached_conversion(d: Path, *, allow_partial: bool = False) -> Path | None:
47
49
  md_path = d / "paper.md"
48
50
  meta_path = d / "meta.json"
49
51
  if not md_path.is_file() or md_path.stat().st_size < MIN_MARKDOWN_CHARS or not meta_path.is_file():
@@ -59,7 +61,9 @@ def _cached_conversion(d: Path, *, allow_native_fallback: bool = False) -> Path
59
61
  meta.get("converter") == CONVERTER
60
62
  and meta.get("bytes") == len(markdown)
61
63
  and meta.get("sha256") == hashlib.sha256(markdown).hexdigest()
62
- and (allow_native_fallback or meta.get("conversion_mode") != "native_fallback")
64
+ and (allow_partial or (meta.get("quality") == "complete" and meta.get("conversion_mode") != "native_fallback"))
65
+ and (d / "paper.pdf").is_file()
66
+ and meta.get("pdf_sha256") == hashlib.sha256((d / "paper.pdf").read_bytes()).hexdigest()
63
67
  )
64
68
  return md_path if valid else None
65
69
 
@@ -167,10 +171,17 @@ def _download_pdf(url: str, pdf_path: Path) -> bool:
167
171
  return True
168
172
 
169
173
 
170
- def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
174
+ def convert(
175
+ arxiv_id: str,
176
+ *,
177
+ allow_partial: bool = False,
178
+ url: str | None = None,
179
+ refresh: bool = False,
180
+ paper_ref: str | None = None,
181
+ ) -> bool:
171
182
  d = CACHE_DIR / arxiv_id
172
- cached = _cached_conversion(d, allow_native_fallback=allow_native_fallback)
173
- if cached is not None:
183
+ cached = _cached_conversion(d, allow_partial=allow_partial)
184
+ if cached is not None and not refresh:
174
185
  print(f"{arxiv_id}: cached at {cached}")
175
186
  return True
176
187
 
@@ -184,10 +195,10 @@ def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
184
195
  )
185
196
  return False
186
197
 
187
- url = f"https://arxiv.org/pdf/{arxiv_id}"
198
+ url = url or f"https://arxiv.org/pdf/{arxiv_id}"
188
199
  d.mkdir(parents=True, exist_ok=True)
189
200
  pdf_path = d / "paper.pdf"
190
- reused_cached_pdf = _is_pdf(pdf_path)
201
+ reused_cached_pdf = not refresh and _is_pdf(pdf_path)
191
202
  if not reused_cached_pdf and not _download_pdf(url, pdf_path):
192
203
  return False
193
204
 
@@ -233,10 +244,21 @@ def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
233
244
  print(f"{arxiv_id}: converted to only {chars} chars, treat as a failure", file=sys.stderr)
234
245
  return False
235
246
 
247
+ if paper_ref:
248
+ meta["paper_ref"] = paper_ref
249
+ if not paper_ref.startswith("arxiv:"):
250
+ meta.pop("arxiv_id", None)
251
+ meta["pdf_sha256"] = hashlib.sha256(pdf_path.read_bytes()).hexdigest()
236
252
  md_path = d / "paper.md"
237
253
  md_path.write_bytes(md.encode("utf-8"))
238
254
  (d / "meta.json").write_text(json.dumps(meta, indent=2), encoding="utf-8")
239
255
  print(f"{arxiv_id}: {len(md)} chars -> {md_path}")
256
+ if meta["quality"] != "complete" and not allow_partial:
257
+ print(
258
+ "Partial extraction retained; inspect meta.json and the PDF, or explicitly use --allow-partial.",
259
+ file=sys.stderr,
260
+ )
261
+ return False
240
262
  return True
241
263
 
242
264
 
@@ -22,6 +22,8 @@ import urllib.request
22
22
  import zlib
23
23
  from pathlib import Path
24
24
 
25
+ from ..tls import arxiv_context
26
+
25
27
  _CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
26
28
  CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
27
29
 
@@ -43,7 +45,7 @@ def _fetch_bytes(url: str, timeout: int = 60) -> bytes | None:
43
45
  )
44
46
  for attempt in range(3):
45
47
  try:
46
- with urllib.request.urlopen(req, timeout=timeout) as resp:
48
+ with urllib.request.urlopen(req, timeout=timeout, context=arxiv_context()) as resp:
47
49
  return resp.read()
48
50
  except urllib.error.HTTPError as e:
49
51
  if e.code == 429:
@@ -323,34 +325,24 @@ def _clean_title(s: str, macros: dict[str, str] | None = None) -> str:
323
325
  return re.sub(r"\s+", " ", s).strip()
324
326
 
325
327
 
326
- def _load_document(arxiv_id: str, refresh: bool) -> tuple[str, list[dict]]:
327
- src = _ensure_source(arxiv_id, refresh)
328
+ def _load_document(arxiv_id: str, refresh: bool, document: str | None = None) -> tuple[str, list[dict]]:
329
+ src = _ensure_source(arxiv_id, refresh).resolve()
328
330
  cands = _tex_candidates(src)
329
- scored = []
330
- for p in cands:
331
- head = _mask_comments(_read(p)[:200_000])
332
- if r"\documentclass" not in head:
333
- continue
334
- flat = _flatten(p, src)
335
- if flat is None:
336
- continue
337
- sections = _find_sections(flat)
338
- scored.append((r"\begin{document}" in head, len(sections), -len(p.parts), flat, sections))
339
- if not scored: # Fall back when no file declares a document class.
340
- for p in cands:
341
- flat = _flatten(p, src)
342
- if flat is None:
343
- continue
344
- sections = _find_sections(flat)
345
- scored.append((False, len(sections), -len(p.parts), flat, sections))
346
- if not scored:
331
+ roots = [p for p in cands if r"\documentclass" in _mask_comments(_read(p)[:200_000])]
332
+ if document:
333
+ selected = (src / document).resolve()
334
+ if not selected.is_relative_to(src.resolve()) or selected not in [p.resolve() for p in cands]:
335
+ sys.exit(f"unknown source document: {document}")
336
+ cands = [selected]
337
+ elif len(roots or cands) > 1:
338
+ names = ", ".join(str(p.relative_to(src)) for p in (roots or cands))
339
+ sys.exit(f"multiple source documents: {names}; select --document (inspect each supplement separately)")
340
+ selected = cands[0] if document else (roots or cands)[0] if cands else None
341
+ flat = _flatten(selected, src) if selected else None
342
+ if flat is None:
347
343
  sys.exit(f"{arxiv_id}: no usable .tex file in {src}")
348
- best = max(scored, key=lambda t: (t[0], t[1], t[2]))
349
- if not best[4]:
350
- sys.exit(f"{arxiv_id}: LaTeX source found but no \\section commands in it")
351
- masked = _mask_comments(best[3])
352
- lo, hi = _body_span(masked)
353
- return best[3][lo:hi], best[4]
344
+ lo, hi = _body_span(_mask_comments(flat))
345
+ return flat[lo:hi], _find_sections(flat)
354
346
 
355
347
 
356
348
  def _load(arxiv_id: str, refresh: bool) -> list[dict]:
@@ -387,7 +379,12 @@ def cmd_section(args: argparse.Namespace) -> None:
387
379
 
388
380
 
389
381
  def cmd_read(args: argparse.Namespace) -> None:
390
- body, sections = _load_document(args.arxiv_id, args.refresh)
382
+ if getattr(args, "documents", False):
383
+ src = _ensure_source(args.arxiv_id, args.refresh)
384
+ for path in _tex_candidates(src):
385
+ print(path.relative_to(src))
386
+ return
387
+ body, sections = _load_document(args.arxiv_id, args.refresh, getattr(args, "document", None))
391
388
  if args.outline:
392
389
  for section in sections:
393
390
  print(f"{section['id']}\t{section['level']}\t{section['title']}")
@@ -426,6 +423,8 @@ def main(argv: list[str]) -> None:
426
423
  mode.add_argument("--section", dest="section_id", help="dotted id (3.2) or exact title")
427
424
  p_read.add_argument("--max-chars", type=int, default=0, help="0 = all remaining text")
428
425
  p_read.add_argument("--start", type=int, default=0)
426
+ p_read.add_argument("--documents", action="store_true")
427
+ p_read.add_argument("--document")
429
428
  p_read.add_argument("--refresh", action="store_true", help="refetch even if cached")
430
429
  p_read.set_defaults(func=cmd_read)
431
430
 
@@ -20,6 +20,7 @@ from pathlib import Path
20
20
  from filelock import FileLock
21
21
 
22
22
  from . import credentials
23
+ from .tls import arxiv_context
23
24
 
24
25
  ARXIV_NS = {"atom": "http://www.w3.org/2005/Atom", "arxiv": "http://arxiv.org/schemas/atom"}
25
26
  SOURCES = ("semantic_scholar", "dblp", "crossref", "openreview", "acl_anthology", "arxiv")
@@ -36,18 +37,21 @@ class PaperRef:
36
37
  value: str
37
38
 
38
39
  @classmethod
39
- def parse(cls, raw: str) -> PaperRef:
40
+ def parse(cls, raw: str, *, preserve_version: bool = False) -> PaperRef:
40
41
  if ":" not in raw:
41
42
  raise ValueError("paper reference needs a prefix: arxiv:, doi:, dblp:, or openreview:")
42
43
  kind, value = raw.strip().split(":", 1)
43
44
  if kind not in ("arxiv", "doi", "dblp", "openreview") or not value:
44
45
  raise ValueError("paper reference needs a prefix: arxiv:, doi:, dblp:, or openreview:")
45
46
  if kind == "arxiv":
46
- value = re.sub(r"v\d+$", "", value)
47
+ versioned = value
48
+ value = re.sub(r"v[1-9]\d*$", "", value)
47
49
  modern = re.fullmatch(r"\d{2}(?:0[1-9]|1[0-2])\.\d{4,5}", value)
48
50
  legacy = re.fullmatch(r"[A-Za-z][A-Za-z.-]*/\d{2}(?:0[1-9]|1[0-2])\d{3}", value)
49
51
  if not (modern or legacy):
50
52
  raise ValueError("invalid arXiv reference")
53
+ if preserve_version:
54
+ value = versioned
51
55
  return cls(kind, value)
52
56
 
53
57
 
@@ -123,6 +127,7 @@ def request(
123
127
  **(headers or {}),
124
128
  }
125
129
  req = urllib.request.Request(url, headers=request_headers, data=data)
130
+ open_options = {"context": arxiv_context()} if "arxiv.org" in host else {}
126
131
  with _request_slot(request_key) as shared_timestamp:
127
132
  for attempt in range(3):
128
133
  elapsed = min(
@@ -132,7 +137,7 @@ def request(
132
137
  if elapsed < interval:
133
138
  time.sleep(interval - elapsed)
134
139
  try:
135
- with urllib.request.urlopen(req, timeout=30) as response:
140
+ with urllib.request.urlopen(req, timeout=30, **open_options) as response:
136
141
  _last_request[request_key] = time.monotonic()
137
142
  _mark_request(shared_timestamp)
138
143
  return response.read()
@@ -0,0 +1,15 @@
1
+ """TLS settings for hosts that reject some client handshakes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import functools
6
+ import ssl
7
+
8
+
9
+ @functools.cache
10
+ def arxiv_context() -> ssl.SSLContext:
11
+ # arXiv's Fastly edge answers uncached requests with 406 when the TLS 1.3
12
+ # ClientHello offers the X25519MLKEM768 hybrid key share (OpenSSL >= 3.5 default).
13
+ context = ssl.create_default_context()
14
+ context.set_ecdh_curve("X25519")
15
+ return context
File without changes