paperstack-cli 0.4.2__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/PKG-INFO +16 -3
  2. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/README.md +15 -2
  3. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/cli.py +73 -38
  4. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/content/arxiv_pdf.py +27 -7
  5. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/content/arxiv_source.py +24 -27
  6. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/metadata.py +5 -2
  7. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/.gitignore +0 -0
  8. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/LICENSE +0 -0
  9. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/pyproject.toml +0 -0
  10. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/scripts/build/site/app.js +0 -0
  11. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/scripts/build/site/entry.html +0 -0
  12. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/scripts/build/site/favicon.svg +0 -0
  13. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/scripts/build/site/index.html +0 -0
  14. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/scripts/build/site/style.css +0 -0
  15. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/scripts/build/vendor/marked.LICENSE +0 -0
  16. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/scripts/build/vendor/marked.min.js +0 -0
  17. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/__init__.py +0 -0
  18. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/arxiv.py +0 -0
  19. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/citations.py +0 -0
  20. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/content/__init__.py +0 -0
  21. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/content/vendor/latexpand +0 -0
  22. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/content/vendor/latexpand.LICENSE +0 -0
  23. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/corpora.py +0 -0
  24. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/credentials.py +0 -0
  25. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/dblp_build.py +0 -0
  26. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/dblp_catalog.py +0 -0
  27. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/dblp_index.py +0 -0
  28. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/entry_types.py +0 -0
  29. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/entrypoint.py +0 -0
  30. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/semantic_scholar.py +0 -0
  31. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/tls.py +0 -0
  32. {paperstack_cli-0.4.2 → paperstack_cli-0.5.0}/src/paperstack/viewer.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: paperstack-cli
3
- Version: 0.4.2
3
+ Version: 0.5.0
4
4
  Summary: Review, inspect, and retrieve research sources from one CLI
5
5
  Project-URL: Repository, https://github.com/MilkClouds/paperstack
6
6
  Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
@@ -146,12 +146,25 @@ paperstack paper read arxiv:2604.23073 --section 6
146
146
  paperstack paper pdf arxiv:2602.09017
147
147
  ```
148
148
 
149
- `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf` require an `arxiv:`
150
- reference. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
149
+ `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
150
+ accept arXiv references directly; other identifiers require an explicit primary-source HTTPS `--pdf-url`. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
151
151
  `pmid:`, and `mag:` identifiers. Structured commands expose scoped `--json` flags; networked commands expose scoped
152
152
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
153
153
  of a work should be cited.
154
154
 
155
+ ### Source reading
156
+
157
+ - Pin `arxiv:IDvN` for version-specific downloads and caches. Metadata lookup remains work-level.
158
+ Unversioned reads may use stale cached content; `--refresh` refetches, `--offline` never does.
159
+ - `read ... --documents` lists source files. Multiple TeX roots require `--document FILE`;
160
+ inspect relevant supplements separately. `--outline`/`--section` apply to that document.
161
+ - DOI/OpenReview/DBLP reads require `--pdf-url https://...` pointing to the primary PDF.
162
+ `read` prints Markdown; `pdf` prints its cache path. New conversions need the PDF extra.
163
+ - Partial PDF extraction returns failure unless `--allow-partial` is explicit, even offline.
164
+ Check `meta.json` and the PDF before using numbers. Converter quality does not certify accuracy.
165
+ - PDF caches verify original and Markdown hashes. Old caches without PDF hashes need rebuilding.
166
+
167
+
155
168
  OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
156
169
  filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
157
170
  public invitation, venue, decision, and status fields rather than treating it as a universal field.
@@ -124,12 +124,25 @@ paperstack paper read arxiv:2604.23073 --section 6
124
124
  paperstack paper pdf arxiv:2602.09017
125
125
  ```
126
126
 
127
- `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf` require an `arxiv:`
128
- reference. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
127
+ `metadata` accepts `arxiv:`, `doi:`, `dblp:`, and `openreview:` references. `read` and `pdf`
128
+ accept arXiv references directly; other identifiers require an explicit primary-source HTTPS `--pdf-url`. `authors`, `citations`, and `references` use Semantic Scholar and also accept its `s2:`, `corpus:`, `acl:`,
129
129
  `pmid:`, and `mag:` identifiers. Structured commands expose scoped `--json` flags; networked commands expose scoped
130
130
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
131
131
  of a work should be cited.
132
132
 
133
+ ### Source reading
134
+
135
+ - Pin `arxiv:IDvN` for version-specific downloads and caches. Metadata lookup remains work-level.
136
+ Unversioned reads may use stale cached content; `--refresh` refetches, `--offline` never does.
137
+ - `read ... --documents` lists source files. Multiple TeX roots require `--document FILE`;
138
+ inspect relevant supplements separately. `--outline`/`--section` apply to that document.
139
+ - DOI/OpenReview/DBLP reads require `--pdf-url https://...` pointing to the primary PDF.
140
+ `read` prints Markdown; `pdf` prints its cache path. New conversions need the PDF extra.
141
+ - Partial PDF extraction returns failure unless `--allow-partial` is explicit, even offline.
142
+ Check `meta.json` and the PDF before using numbers. Converter quality does not certify accuracy.
143
+ - PDF caches verify original and Markdown hashes. Old caches without PDF hashes need rebuilding.
144
+
145
+
133
146
  OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
134
147
  filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
135
148
  public invitation, venue, decision, and status fields rather than treating it as a universal field.
@@ -872,6 +872,43 @@ def _paper_cache() -> Path:
872
872
  return Path(os.environ.get("PAPERSTACK_PAPERS_DIR", base / "paperstack" / "papers"))
873
873
 
874
874
 
875
+ def _read_pdf(a, ref, offline):
876
+ import hashlib
877
+ from contextlib import redirect_stdout
878
+ from urllib.parse import urlparse
879
+
880
+ from .content import arxiv_pdf
881
+ from .content.arxiv_source import _print_chunk
882
+
883
+ url = a.pdf_url
884
+ if ref.kind != "arxiv" and not url:
885
+ die("non-arXiv content requires --pdf-url with the primary-source PDF URL")
886
+ if url and (urlparse(url).scheme != "https" or not urlparse(url).hostname or urlparse(url).username):
887
+ die("--pdf-url must be an HTTPS URL without credentials")
888
+ if offline and a.refresh:
889
+ die("--offline and --refresh cannot be used together")
890
+ if a.paper_cmd == "read" and (a.outline or a.section_id or a.documents or a.document):
891
+ die("PDF reads support --start/--max-chars, not LaTeX document or section selection")
892
+ key = ref.value if not url else "external-" + hashlib.sha256(f"{ref.kind}:{ref.value}\n{url}".encode()).hexdigest()
893
+ arxiv_pdf.CACHE_DIR = _paper_cache()
894
+ directory = arxiv_pdf.CACHE_DIR / key
895
+ if offline:
896
+ if arxiv_pdf._cached_conversion(directory, allow_partial=a.allow_partial) is None:
897
+ die("no usable cached PDF conversion; partial content requires --allow-partial")
898
+ else:
899
+ with redirect_stdout(sys.stderr):
900
+ converted = arxiv_pdf.convert(
901
+ key, url=url, refresh=a.refresh, allow_partial=a.allow_partial, paper_ref=f"{ref.kind}:{ref.value}"
902
+ )
903
+ if not converted:
904
+ return 1
905
+ if a.paper_cmd == "read":
906
+ _print_chunk((directory / "paper.md").read_text(), a.start, a.max_chars)
907
+ else:
908
+ print(directory / "paper.md")
909
+ return 0
910
+
911
+
875
912
  def _run_paper(a: argparse.Namespace) -> int:
876
913
  from . import credentials, metadata
877
914
 
@@ -1008,44 +1045,35 @@ def _run_paper(a: argparse.Namespace) -> int:
1008
1045
  return 0 if result["status"] == "ok" else 1
1009
1046
 
1010
1047
  try:
1011
- ref = metadata.PaperRef.parse(a.paper_ref)
1048
+ ref = metadata.PaperRef.parse(a.paper_ref, preserve_version=True)
1012
1049
  except ValueError as exc:
1013
1050
  die(str(exc))
1014
- if ref.kind != "arxiv":
1015
- die(f"paper {a.paper_cmd} currently requires an arxiv: reference")
1016
- if a.paper_cmd == "read":
1017
- from .content import arxiv_source
1018
-
1019
- arxiv_source.CACHE_DIR = _paper_cache()
1020
- cached_source = arxiv_source.CACHE_DIR / ref.value / "src"
1021
- if offline and a.refresh:
1022
- die("--offline and --refresh cannot be used together")
1023
- if offline and (not cached_source.is_dir() or not arxiv_source._tex_candidates(cached_source)):
1024
- die(f"no complete cached source for arxiv:{ref.value}")
1025
- argv = ["read", ref.value]
1026
- if a.outline:
1027
- argv.append("--outline")
1028
- elif a.section_id:
1029
- argv.extend(["--section", a.section_id])
1030
- if a.refresh:
1031
- argv.append("--refresh")
1032
- argv.extend(["--start", str(a.start), "--max-chars", str(a.max_chars)])
1033
- arxiv_source.main(argv)
1034
- return 0
1035
- if a.paper_cmd == "pdf":
1036
- from .content import arxiv_pdf
1037
-
1038
- arxiv_pdf.CACHE_DIR = _paper_cache()
1039
- cached_pdf = arxiv_pdf._cached_conversion(
1040
- arxiv_pdf.CACHE_DIR / ref.value,
1041
- allow_native_fallback=offline,
1042
- )
1043
- if offline and cached_pdf is None:
1044
- die(f"no usable cached PDF conversion for arxiv:{ref.value}")
1045
- if not arxiv_pdf.convert(ref.value, allow_native_fallback=offline):
1046
- return 1
1047
- return 0
1048
- raise AssertionError(a.paper_cmd)
1051
+ if ref.kind == "arxiv" and not re.search(r"v[1-9]\d*$", ref.value):
1052
+ warn("Unversioned arXiv reference: cached content may not be current; pin vN for reproducible reads.")
1053
+ if a.pdf_url or a.paper_cmd == "pdf" or ref.kind != "arxiv":
1054
+ return _read_pdf(a, ref, offline)
1055
+ from .content import arxiv_source
1056
+
1057
+ arxiv_source.CACHE_DIR = _paper_cache()
1058
+ cached_source = arxiv_source.CACHE_DIR / ref.value / "src"
1059
+ if offline and a.refresh:
1060
+ die("--offline and --refresh cannot be used together")
1061
+ if offline and (not cached_source.is_dir() or not arxiv_source._tex_candidates(cached_source)):
1062
+ die(f"no complete cached source for arxiv:{ref.value}")
1063
+ argv = ["read", ref.value]
1064
+ if a.outline:
1065
+ argv.append("--outline")
1066
+ elif a.section_id:
1067
+ argv.extend(["--section", a.section_id])
1068
+ if a.documents:
1069
+ argv.append("--documents")
1070
+ if a.document:
1071
+ argv.extend(["--document", a.document])
1072
+ if a.refresh:
1073
+ argv.append("--refresh")
1074
+ argv.extend(["--start", str(a.start), "--max-chars", str(a.max_chars)])
1075
+ arxiv_source.main(argv)
1076
+ return 0
1049
1077
 
1050
1078
 
1051
1079
  def _run_index(a: argparse.Namespace) -> int:
@@ -1226,16 +1254,23 @@ Use `paperstack review ...` to find or read an authored critical judgment.""",
1226
1254
  _output(s)
1227
1255
  _offline(s)
1228
1256
  s = paper_sub.add_parser("read", help="read the LaTeX body, outline, or one section")
1229
- s.add_argument("paper_ref", help="arxiv: reference")
1257
+ s.add_argument("paper_ref", help="arxiv: (version preserved), doi:, dblp:, or openreview: reference")
1230
1258
  mode = s.add_mutually_exclusive_group()
1231
1259
  mode.add_argument("--outline", action="store_true", help="print numbered section headings only")
1232
1260
  mode.add_argument("--section", dest="section_id", help="print one section by outline number")
1233
1261
  s.add_argument("--refresh", action="store_true", help="replace the cached arXiv source")
1234
1262
  s.add_argument("--start", type=int, default=0, help="start at this character offset")
1235
1263
  s.add_argument("--max-chars", type=int, default=0, help="truncate output after this many characters")
1264
+ s.add_argument("--documents", action="store_true", help="list source files, including supplements")
1265
+ s.add_argument("--document", help="source-relative TeX root to read")
1266
+ s.add_argument("--pdf-url", help="explicit primary-source HTTPS PDF URL")
1267
+ s.add_argument("--allow-partial", action="store_true", help="explicitly accept incomplete PDF extraction")
1236
1268
  _offline(s)
1237
1269
  s = paper_sub.add_parser("pdf", help="download and convert a native PDF submission")
1238
- s.add_argument("paper_ref", help="arxiv: reference")
1270
+ s.add_argument("paper_ref", help="arxiv: (version preserved), doi:, dblp:, or openreview: reference")
1271
+ s.add_argument("--pdf-url", help="explicit primary-source HTTPS PDF URL")
1272
+ s.add_argument("--allow-partial", action="store_true", help="explicitly accept incomplete PDF extraction")
1273
+ s.add_argument("--refresh", action="store_true", help="download and convert the PDF again")
1239
1274
  _offline(s)
1240
1275
 
1241
1276
  index = sub.add_parser("index", help="optional local lookup indexes")
@@ -45,7 +45,7 @@ def _fetch(url: str, timeout: int = 60) -> bytes | None:
45
45
  return None
46
46
 
47
47
 
48
- def _cached_conversion(d: Path, *, allow_native_fallback: bool = False) -> Path | None:
48
+ def _cached_conversion(d: Path, *, allow_partial: bool = False) -> Path | None:
49
49
  md_path = d / "paper.md"
50
50
  meta_path = d / "meta.json"
51
51
  if not md_path.is_file() or md_path.stat().st_size < MIN_MARKDOWN_CHARS or not meta_path.is_file():
@@ -61,7 +61,9 @@ def _cached_conversion(d: Path, *, allow_native_fallback: bool = False) -> Path
61
61
  meta.get("converter") == CONVERTER
62
62
  and meta.get("bytes") == len(markdown)
63
63
  and meta.get("sha256") == hashlib.sha256(markdown).hexdigest()
64
- and (allow_native_fallback or meta.get("conversion_mode") != "native_fallback")
64
+ and (allow_partial or (meta.get("quality") == "complete" and meta.get("conversion_mode") != "native_fallback"))
65
+ and (d / "paper.pdf").is_file()
66
+ and meta.get("pdf_sha256") == hashlib.sha256((d / "paper.pdf").read_bytes()).hexdigest()
65
67
  )
66
68
  return md_path if valid else None
67
69
 
@@ -169,10 +171,17 @@ def _download_pdf(url: str, pdf_path: Path) -> bool:
169
171
  return True
170
172
 
171
173
 
172
- def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
174
+ def convert(
175
+ arxiv_id: str,
176
+ *,
177
+ allow_partial: bool = False,
178
+ url: str | None = None,
179
+ refresh: bool = False,
180
+ paper_ref: str | None = None,
181
+ ) -> bool:
173
182
  d = CACHE_DIR / arxiv_id
174
- cached = _cached_conversion(d, allow_native_fallback=allow_native_fallback)
175
- if cached is not None:
183
+ cached = _cached_conversion(d, allow_partial=allow_partial)
184
+ if cached is not None and not refresh:
176
185
  print(f"{arxiv_id}: cached at {cached}")
177
186
  return True
178
187
 
@@ -186,10 +195,10 @@ def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
186
195
  )
187
196
  return False
188
197
 
189
- url = f"https://arxiv.org/pdf/{arxiv_id}"
198
+ url = url or f"https://arxiv.org/pdf/{arxiv_id}"
190
199
  d.mkdir(parents=True, exist_ok=True)
191
200
  pdf_path = d / "paper.pdf"
192
- reused_cached_pdf = _is_pdf(pdf_path)
201
+ reused_cached_pdf = not refresh and _is_pdf(pdf_path)
193
202
  if not reused_cached_pdf and not _download_pdf(url, pdf_path):
194
203
  return False
195
204
 
@@ -235,10 +244,21 @@ def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
235
244
  print(f"{arxiv_id}: converted to only {chars} chars, treat as a failure", file=sys.stderr)
236
245
  return False
237
246
 
247
+ if paper_ref:
248
+ meta["paper_ref"] = paper_ref
249
+ if not paper_ref.startswith("arxiv:"):
250
+ meta.pop("arxiv_id", None)
251
+ meta["pdf_sha256"] = hashlib.sha256(pdf_path.read_bytes()).hexdigest()
238
252
  md_path = d / "paper.md"
239
253
  md_path.write_bytes(md.encode("utf-8"))
240
254
  (d / "meta.json").write_text(json.dumps(meta, indent=2), encoding="utf-8")
241
255
  print(f"{arxiv_id}: {len(md)} chars -> {md_path}")
256
+ if meta["quality"] != "complete" and not allow_partial:
257
+ print(
258
+ "Partial extraction retained; inspect meta.json and the PDF, or explicitly use --allow-partial.",
259
+ file=sys.stderr,
260
+ )
261
+ return False
242
262
  return True
243
263
 
244
264
 
@@ -325,34 +325,24 @@ def _clean_title(s: str, macros: dict[str, str] | None = None) -> str:
325
325
  return re.sub(r"\s+", " ", s).strip()
326
326
 
327
327
 
328
- def _load_document(arxiv_id: str, refresh: bool) -> tuple[str, list[dict]]:
329
- src = _ensure_source(arxiv_id, refresh)
328
+ def _load_document(arxiv_id: str, refresh: bool, document: str | None = None) -> tuple[str, list[dict]]:
329
+ src = _ensure_source(arxiv_id, refresh).resolve()
330
330
  cands = _tex_candidates(src)
331
- scored = []
332
- for p in cands:
333
- head = _mask_comments(_read(p)[:200_000])
334
- if r"\documentclass" not in head:
335
- continue
336
- flat = _flatten(p, src)
337
- if flat is None:
338
- continue
339
- sections = _find_sections(flat)
340
- scored.append((r"\begin{document}" in head, len(sections), -len(p.parts), flat, sections))
341
- if not scored: # Fall back when no file declares a document class.
342
- for p in cands:
343
- flat = _flatten(p, src)
344
- if flat is None:
345
- continue
346
- sections = _find_sections(flat)
347
- scored.append((False, len(sections), -len(p.parts), flat, sections))
348
- if not scored:
331
+ roots = [p for p in cands if r"\documentclass" in _mask_comments(_read(p)[:200_000])]
332
+ if document:
333
+ selected = (src / document).resolve()
334
+ if not selected.is_relative_to(src.resolve()) or selected not in [p.resolve() for p in cands]:
335
+ sys.exit(f"unknown source document: {document}")
336
+ cands = [selected]
337
+ elif len(roots or cands) > 1:
338
+ names = ", ".join(str(p.relative_to(src)) for p in (roots or cands))
339
+ sys.exit(f"multiple source documents: {names}; select --document (inspect each supplement separately)")
340
+ selected = cands[0] if document else (roots or cands)[0] if cands else None
341
+ flat = _flatten(selected, src) if selected else None
342
+ if flat is None:
349
343
  sys.exit(f"{arxiv_id}: no usable .tex file in {src}")
350
- best = max(scored, key=lambda t: (t[0], t[1], t[2]))
351
- if not best[4]:
352
- sys.exit(f"{arxiv_id}: LaTeX source found but no \\section commands in it")
353
- masked = _mask_comments(best[3])
354
- lo, hi = _body_span(masked)
355
- return best[3][lo:hi], best[4]
344
+ lo, hi = _body_span(_mask_comments(flat))
345
+ return flat[lo:hi], _find_sections(flat)
356
346
 
357
347
 
358
348
  def _load(arxiv_id: str, refresh: bool) -> list[dict]:
@@ -389,7 +379,12 @@ def cmd_section(args: argparse.Namespace) -> None:
389
379
 
390
380
 
391
381
  def cmd_read(args: argparse.Namespace) -> None:
392
- body, sections = _load_document(args.arxiv_id, args.refresh)
382
+ if getattr(args, "documents", False):
383
+ src = _ensure_source(args.arxiv_id, args.refresh)
384
+ for path in _tex_candidates(src):
385
+ print(path.relative_to(src))
386
+ return
387
+ body, sections = _load_document(args.arxiv_id, args.refresh, getattr(args, "document", None))
393
388
  if args.outline:
394
389
  for section in sections:
395
390
  print(f"{section['id']}\t{section['level']}\t{section['title']}")
@@ -428,6 +423,8 @@ def main(argv: list[str]) -> None:
428
423
  mode.add_argument("--section", dest="section_id", help="dotted id (3.2) or exact title")
429
424
  p_read.add_argument("--max-chars", type=int, default=0, help="0 = all remaining text")
430
425
  p_read.add_argument("--start", type=int, default=0)
426
+ p_read.add_argument("--documents", action="store_true")
427
+ p_read.add_argument("--document")
431
428
  p_read.add_argument("--refresh", action="store_true", help="refetch even if cached")
432
429
  p_read.set_defaults(func=cmd_read)
433
430
 
@@ -37,18 +37,21 @@ class PaperRef:
37
37
  value: str
38
38
 
39
39
  @classmethod
40
- def parse(cls, raw: str) -> PaperRef:
40
+ def parse(cls, raw: str, *, preserve_version: bool = False) -> PaperRef:
41
41
  if ":" not in raw:
42
42
  raise ValueError("paper reference needs a prefix: arxiv:, doi:, dblp:, or openreview:")
43
43
  kind, value = raw.strip().split(":", 1)
44
44
  if kind not in ("arxiv", "doi", "dblp", "openreview") or not value:
45
45
  raise ValueError("paper reference needs a prefix: arxiv:, doi:, dblp:, or openreview:")
46
46
  if kind == "arxiv":
47
- value = re.sub(r"v\d+$", "", value)
47
+ versioned = value
48
+ value = re.sub(r"v[1-9]\d*$", "", value)
48
49
  modern = re.fullmatch(r"\d{2}(?:0[1-9]|1[0-2])\.\d{4,5}", value)
49
50
  legacy = re.fullmatch(r"[A-Za-z][A-Za-z.-]*/\d{2}(?:0[1-9]|1[0-2])\d{3}", value)
50
51
  if not (modern or legacy):
51
52
  raise ValueError("invalid arXiv reference")
53
+ if preserve_version:
54
+ value = versioned
52
55
  return cls(kind, value)
53
56
 
54
57
 
File without changes