paperstack-cli 0.3.2__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. paperstack_cli-0.3.2/README.md → paperstack_cli-0.4.1/PKG-INFO +41 -0
  2. paperstack_cli-0.3.2/PKG-INFO → paperstack_cli-0.4.1/README.md +19 -17
  3. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/pyproject.toml +8 -1
  4. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/cli.py +62 -21
  5. paperstack_cli-0.4.1/src/paperstack/content/arxiv_pdf.py +253 -0
  6. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/entrypoint.py +5 -0
  7. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/metadata.py +192 -45
  8. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/viewer.py +3 -2
  9. paperstack_cli-0.3.2/src/paperstack/content/arxiv_pdf.py +0 -103
  10. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/.gitignore +0 -0
  11. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/LICENSE +0 -0
  12. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/scripts/build/site/app.js +0 -0
  13. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/scripts/build/site/entry.html +0 -0
  14. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/scripts/build/site/favicon.svg +0 -0
  15. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/scripts/build/site/index.html +0 -0
  16. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/scripts/build/site/style.css +0 -0
  17. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/scripts/build/vendor/marked.LICENSE +0 -0
  18. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/scripts/build/vendor/marked.min.js +0 -0
  19. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/__init__.py +0 -0
  20. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/arxiv.py +0 -0
  21. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/citations.py +0 -0
  22. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/content/__init__.py +0 -0
  23. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/content/arxiv_source.py +0 -0
  24. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/content/vendor/latexpand +0 -0
  25. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/content/vendor/latexpand.LICENSE +0 -0
  26. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/corpora.py +0 -0
  27. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/credentials.py +0 -0
  28. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/dblp_build.py +0 -0
  29. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/dblp_catalog.py +0 -0
  30. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/dblp_index.py +0 -0
  31. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/entry_types.py +0 -0
  32. {paperstack_cli-0.3.2 → paperstack_cli-0.4.1}/src/paperstack/semantic_scholar.py +0 -0
@@ -1,5 +1,29 @@
1
+ Metadata-Version: 2.5
2
+ Name: paperstack-cli
3
+ Version: 0.4.1
4
+ Summary: Review, inspect, and retrieve research sources from one CLI
5
+ Project-URL: Repository, https://github.com/MilkClouds/paperstack
6
+ Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
7
+ License-Expression: Apache-2.0
8
+ License-File: LICENSE
9
+ Classifier: Programming Language :: Python :: 3 :: Only
10
+ Classifier: Programming Language :: Python :: 3.11
11
+ Classifier: Programming Language :: Python :: 3.12
12
+ Classifier: Programming Language :: Python :: 3.13
13
+ Classifier: Programming Language :: Python :: 3.14
14
+ Requires-Python: >=3.11
15
+ Requires-Dist: filelock>=3.20
16
+ Requires-Dist: polars>=1.43
17
+ Requires-Dist: python-dotenv>=1.2.2
18
+ Requires-Dist: pyyaml>=6
19
+ Provides-Extra: pdf
20
+ Requires-Dist: pdf-inspector<2,>=1.17; extra == 'pdf'
21
+ Description-Content-Type: text/markdown
22
+
1
23
  # Paperstack
2
24
 
25
+ [![CI](https://github.com/MilkClouds/paperstack/actions/workflows/ci.yml/badge.svg)](https://github.com/MilkClouds/paperstack/actions/workflows/ci.yml) [![PyPI version](https://img.shields.io/pypi/v/paperstack-cli.svg)](https://pypi.org/project/paperstack-cli/) [![Python versions](https://img.shields.io/pypi/pyversions/paperstack-cli.svg)](https://pypi.org/project/paperstack-cli/) [![License](https://img.shields.io/pypi/l/paperstack-cli.svg)](https://github.com/MilkClouds/paperstack/blob/main/LICENSE)
26
+
3
27
  Paperstack is an open-source CLI for maintaining, querying, and publishing an independently authored research corpus.
4
28
  The software and corpus are separate: Paperstack can be public while each corpus and its editorial judgments remain
5
29
  under its maintainer's control.
@@ -21,6 +45,18 @@ PDF conversion is optional:
21
45
  uv tool install 'paperstack-cli[pdf]'
22
46
  ```
23
47
 
48
+ The PDF extra uses `pdf-inspector` for native extraction and selective OCR.
49
+ Clean text PDFs need no external runtime, while routed OCR pages require the pinned PDFium and ONNX Runtime libraries described in the [OCR runtime guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md).
50
+ The first routed page downloads and verifies the pinned OCR model set unless an offline model directory is configured.
51
+ When the shared libraries are not on the platform search path, point `pdf-inspector` at the extracted files:
52
+
53
+ ```bash
54
+ export PDFIUM_LIB_PATH=/absolute/path/to/libpdfium.so
55
+ export ORT_DYLIB_PATH=/absolute/path/to/libonnxruntime.so
56
+ ```
57
+
58
+ The runtime guide lists the matching downloads and filenames for Linux, macOS, and Windows.
59
+
24
60
  ## AI agent skill
25
61
 
26
62
  Paperstack ships an [Agent Skills](https://agentskills.io)-compatible skill for Codex, Claude Code, and other
@@ -98,6 +134,7 @@ These commands use external source records and do not select a citation or make
98
134
 
99
135
  ```bash
100
136
  paperstack paper search "Attention Is All You Need" --source dblp
137
+ paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
101
138
  paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
102
139
  paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
103
140
  paperstack paper metadata arxiv:2106.09685
@@ -115,6 +152,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
115
152
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
116
153
  of a work should be cited.
117
154
 
155
+ OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
156
+ filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
157
+ public invitation, venue, decision, and status fields rather than treating it as a universal field.
158
+
118
159
  ## Build a viewer
119
160
 
120
161
  ```bash
@@ -1,22 +1,7 @@
1
- Metadata-Version: 2.5
2
- Name: paperstack-cli
3
- Version: 0.3.2
4
- Summary: Review, inspect, and retrieve research sources from one CLI
5
- Project-URL: Repository, https://github.com/MilkClouds/paperstack
6
- Project-URL: Issues, https://github.com/MilkClouds/paperstack/issues
7
- License-Expression: Apache-2.0
8
- License-File: LICENSE
9
- Requires-Python: >=3.11
10
- Requires-Dist: filelock>=3.20
11
- Requires-Dist: polars>=1.43
12
- Requires-Dist: python-dotenv>=1.2.2
13
- Requires-Dist: pyyaml>=6
14
- Provides-Extra: pdf
15
- Requires-Dist: pymupdf4llm>=0.0.17; extra == 'pdf'
16
- Description-Content-Type: text/markdown
17
-
18
1
  # Paperstack
19
2
 
3
+ [![CI](https://github.com/MilkClouds/paperstack/actions/workflows/ci.yml/badge.svg)](https://github.com/MilkClouds/paperstack/actions/workflows/ci.yml) [![PyPI version](https://img.shields.io/pypi/v/paperstack-cli.svg)](https://pypi.org/project/paperstack-cli/) [![Python versions](https://img.shields.io/pypi/pyversions/paperstack-cli.svg)](https://pypi.org/project/paperstack-cli/) [![License](https://img.shields.io/pypi/l/paperstack-cli.svg)](https://github.com/MilkClouds/paperstack/blob/main/LICENSE)
4
+
20
5
  Paperstack is an open-source CLI for maintaining, querying, and publishing an independently authored research corpus.
21
6
  The software and corpus are separate: Paperstack can be public while each corpus and its editorial judgments remain
22
7
  under its maintainer's control.
@@ -38,6 +23,18 @@ PDF conversion is optional:
38
23
  uv tool install 'paperstack-cli[pdf]'
39
24
  ```
40
25
 
26
+ The PDF extra uses `pdf-inspector` for native extraction and selective OCR.
27
+ Clean text PDFs need no external runtime, while routed OCR pages require the pinned PDFium and ONNX Runtime libraries described in the [OCR runtime guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md).
28
+ The first routed page downloads and verifies the pinned OCR model set unless an offline model directory is configured.
29
+ When the shared libraries are not on the platform search path, point `pdf-inspector` at the extracted files:
30
+
31
+ ```bash
32
+ export PDFIUM_LIB_PATH=/absolute/path/to/libpdfium.so
33
+ export ORT_DYLIB_PATH=/absolute/path/to/libonnxruntime.so
34
+ ```
35
+
36
+ The runtime guide lists the matching downloads and filenames for Linux, macOS, and Windows.
37
+
41
38
  ## AI agent skill
42
39
 
43
40
  Paperstack ships an [Agent Skills](https://agentskills.io)-compatible skill for Codex, Claude Code, and other
@@ -115,6 +112,7 @@ These commands use external source records and do not select a citation or make
115
112
 
116
113
  ```bash
117
114
  paperstack paper search "Attention Is All You Need" --source dblp
115
+ paperstack paper search "Exact Paper Title" --source openreview --exact-title --openreview-status accepted
118
116
  paperstack paper search "robot learning" --source semantic-scholar --year 2024-2026
119
117
  paperstack paper search 'ti:"robot learning"' --source arxiv --category cs.RO --sort date
120
118
  paperstack paper metadata arxiv:2106.09685
@@ -132,6 +130,10 @@ reference. `authors`, `citations`, and `references` use Semantic Scholar and als
132
130
  `--offline` flags. Paperstack reports source records but does not synthesize a citation entry or choose which version
133
131
  of a work should be cited.
134
132
 
133
+ OpenReview exact-title search uses its title-only exact mode and verifies a normalized title match locally. Status
134
+ filtering is conservative because venues encode decisions and withdrawals differently; Paperstack infers it from
135
+ public invitation, venue, decision, and status fields rather than treating it as a universal field.
136
+
135
137
  ## Build a viewer
136
138
 
137
139
  ```bash
@@ -5,6 +5,13 @@ description = "Review, inspect, and retrieve research sources from one CLI"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
7
7
  requires-python = ">=3.11"
8
+ classifiers = [
9
+ "Programming Language :: Python :: 3 :: Only",
10
+ "Programming Language :: Python :: 3.11",
11
+ "Programming Language :: Python :: 3.12",
12
+ "Programming Language :: Python :: 3.13",
13
+ "Programming Language :: Python :: 3.14",
14
+ ]
8
15
  dependencies = [
9
16
  "filelock>=3.20",
10
17
  "polars>=1.43",
@@ -13,7 +20,7 @@ dependencies = [
13
20
  ]
14
21
 
15
22
  [project.optional-dependencies]
16
- pdf = ["pymupdf4llm>=0.0.17"]
23
+ pdf = ["pdf-inspector>=1.17,<2"]
17
24
 
18
25
  [dependency-groups]
19
26
  lint = ["ruff>=0.12"]
@@ -925,12 +925,42 @@ def _run_paper(a: argparse.Namespace) -> int:
925
925
  metadata.print_results(results, json_output=a.json)
926
926
  return 0 if any(item["status"] == "ok" for item in results) else 1
927
927
  if a.paper_cmd == "search":
928
+ semantic_filters = [
929
+ flag
930
+ for flag, selected in (
931
+ ("--offset", a.offset),
932
+ ("--year", a.year),
933
+ ("--field-of-study", a.fields_of_study),
934
+ ("--open-access", a.open_access),
935
+ )
936
+ if selected
937
+ ]
938
+ arxiv_filters = [
939
+ flag
940
+ for flag, selected in (
941
+ ("--category", a.categories),
942
+ ("--date-from", a.date_from),
943
+ ("--date-to", a.date_to),
944
+ ("--sort", a.sort != "relevance"),
945
+ )
946
+ if selected
947
+ ]
948
+ openreview_filters = [
949
+ flag
950
+ for flag, selected in (
951
+ ("--exact-title", a.exact_title),
952
+ ("--openreview-status", a.openreview_status),
953
+ )
954
+ if selected
955
+ ]
928
956
  try:
957
+ if a.source != "openreview" and openreview_filters:
958
+ die(f"--source openreview is required for {', '.join(openreview_filters)}")
929
959
  if a.source == "arxiv":
930
960
  from . import arxiv
931
961
 
932
- if a.year or a.fields_of_study or a.open_access or a.offset:
933
- die("--year, --field-of-study, --open-access, and --offset require --source semantic-scholar")
962
+ if semantic_filters:
963
+ die(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
934
964
  result = arxiv.search(
935
965
  a.query,
936
966
  categories=a.categories,
@@ -942,8 +972,8 @@ def _run_paper(a: argparse.Namespace) -> int:
942
972
  elif a.source == "semantic-scholar":
943
973
  from . import semantic_scholar
944
974
 
945
- if a.categories or a.date_from or a.date_to or a.sort != "relevance":
946
- die("--category, --date-from, --date-to, and --sort require --source arxiv")
975
+ if arxiv_filters:
976
+ die(f"--source arxiv is required for {', '.join(arxiv_filters)}")
947
977
  result = semantic_scholar.search(
948
978
  a.query,
949
979
  limit=a.limit,
@@ -953,19 +983,21 @@ def _run_paper(a: argparse.Namespace) -> int:
953
983
  open_access=a.open_access,
954
984
  )
955
985
  else:
956
- if (
957
- a.categories
958
- or a.date_from
959
- or a.date_to
960
- or a.limit != 10
961
- or a.offset
962
- or a.sort != "relevance"
963
- or a.year
964
- or a.fields_of_study
965
- or a.open_access
966
- ):
967
- die("search filters require --source arxiv or --source semantic-scholar")
968
- result = metadata.search(a.source, a.query, local_only=offline)
986
+ requirements = []
987
+ if semantic_filters:
988
+ requirements.append(f"--source semantic-scholar is required for {', '.join(semantic_filters)}")
989
+ if arxiv_filters:
990
+ requirements.append(f"--source arxiv is required for {', '.join(arxiv_filters)}")
991
+ if requirements:
992
+ die("; ".join(requirements))
993
+ result = metadata.search(
994
+ a.source,
995
+ a.query,
996
+ limit=a.limit,
997
+ local_only=offline,
998
+ exact_title=a.exact_title,
999
+ openreview_status=a.openreview_status,
1000
+ )
969
1001
  except credentials.CredentialsError as exc:
970
1002
  die(f"configuration failed: {exc}")
971
1003
  except RuntimeError as exc:
@@ -1004,10 +1036,13 @@ def _run_paper(a: argparse.Namespace) -> int:
1004
1036
  from .content import arxiv_pdf
1005
1037
 
1006
1038
  arxiv_pdf.CACHE_DIR = _paper_cache()
1007
- cached_pdf = arxiv_pdf.CACHE_DIR / ref.value / "paper.md"
1008
- if offline and (not cached_pdf.is_file() or cached_pdf.stat().st_size <= 1000):
1009
- die(f"no complete cached PDF conversion for arxiv:{ref.value}")
1010
- if not arxiv_pdf.convert(ref.value):
1039
+ cached_pdf = arxiv_pdf._cached_conversion(
1040
+ arxiv_pdf.CACHE_DIR / ref.value,
1041
+ allow_native_fallback=offline,
1042
+ )
1043
+ if offline and cached_pdf is None:
1044
+ die(f"no usable cached PDF conversion for arxiv:{ref.value}")
1045
+ if not arxiv_pdf.convert(ref.value, allow_native_fallback=offline):
1011
1046
  return 1
1012
1047
  return 0
1013
1048
  raise AssertionError(a.paper_cmd)
@@ -1175,6 +1210,12 @@ Use `paperstack review ...` to find or read an authored critical judgment.""",
1175
1210
  s.add_argument("--date-from", help="earliest arXiv submission date")
1176
1211
  s.add_argument("--date-to", help="latest arXiv submission date")
1177
1212
  s.add_argument("--sort", choices=("relevance", "date"), default="relevance")
1213
+ s.add_argument("--exact-title", action="store_true", help="require a normalized exact OpenReview title")
1214
+ s.add_argument(
1215
+ "--openreview-status",
1216
+ choices=("submission", "accepted", "withdrawn"),
1217
+ help="filter OpenReview forum records by inferred status",
1218
+ )
1178
1219
  _output(s)
1179
1220
  _offline(s)
1180
1221
  for command in ("authors", "citations", "references"):
@@ -0,0 +1,253 @@
1
+ """Convert arXiv PDF submissions to Markdown with selective OCR.
2
+
3
+ Used by `paperstack paper pdf` after the source fetcher reports no TeX.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import hashlib
9
+ import importlib.metadata
10
+ import json
11
+ import os
12
+ import sys
13
+ import tempfile
14
+ import time
15
+ import urllib.error
16
+ import urllib.request
17
+ from pathlib import Path
18
+
19
+ _CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
20
+ CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
21
+ CONVERTER = "pdf-inspector"
22
+ MIN_MARKDOWN_CHARS = 100
23
+ OCR_RUNTIME_GUIDE = "https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md"
24
+
25
+
26
+ def _fetch(url: str, timeout: int = 60) -> bytes | None:
27
+ req = urllib.request.Request(url, headers={"User-Agent": "paperstack/1.0 (+arxiv pdf fetch)"})
28
+ for attempt in range(3):
29
+ try:
30
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
31
+ return resp.read()
32
+ except urllib.error.HTTPError as e:
33
+ if e.code == 429:
34
+ time.sleep(15 * (attempt + 1))
35
+ continue
36
+ if attempt == 2:
37
+ return None
38
+ time.sleep(5)
39
+ except (urllib.error.URLError, OSError, TimeoutError):
40
+ if attempt == 2:
41
+ return None
42
+ time.sleep(5)
43
+ return None
44
+
45
+
46
+ def _cached_conversion(d: Path, *, allow_native_fallback: bool = False) -> Path | None:
47
+ md_path = d / "paper.md"
48
+ meta_path = d / "meta.json"
49
+ if not md_path.is_file() or md_path.stat().st_size < MIN_MARKDOWN_CHARS or not meta_path.is_file():
50
+ return None
51
+ try:
52
+ markdown = md_path.read_bytes()
53
+ meta = json.loads(meta_path.read_text(encoding="utf-8"))
54
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError):
55
+ return None
56
+ if not isinstance(meta, dict):
57
+ return None
58
+ valid = (
59
+ meta.get("converter") == CONVERTER
60
+ and meta.get("bytes") == len(markdown)
61
+ and meta.get("sha256") == hashlib.sha256(markdown).hexdigest()
62
+ and (allow_native_fallback or meta.get("conversion_mode") != "native_fallback")
63
+ )
64
+ return md_path if valid else None
65
+
66
+
67
+ def _converter_version() -> str:
68
+ try:
69
+ return importlib.metadata.version(CONVERTER)
70
+ except importlib.metadata.PackageNotFoundError:
71
+ return "unknown"
72
+
73
+
74
+ def _ocr_reasons(items) -> list[dict]:
75
+ return [{"page": item.page, "reasons": list(item.reasons)} for item in items]
76
+
77
+
78
+ def _page_provenance(page) -> dict:
79
+ provenance = page.provenance
80
+ model = provenance.ocr_model
81
+ return {
82
+ "page": page.page_number,
83
+ "source": provenance.source,
84
+ "ocr_model": None if model is None else {"name": model.name, "revision": model.revision},
85
+ "render_dpi": provenance.render_dpi,
86
+ "ocr_confidence": provenance.ocr_confidence,
87
+ "hosted_recommended": provenance.hosted_recommended,
88
+ "warnings": list(provenance.warnings),
89
+ "timings": {
90
+ "render_ms": provenance.timings.render_ms,
91
+ "ocr_ms": provenance.timings.ocr_ms,
92
+ "assembly_ms": provenance.timings.assembly_ms,
93
+ },
94
+ }
95
+
96
+
97
+ def _base_meta(arxiv_id: str, url: str, markdown: str) -> dict:
98
+ encoded = markdown.encode("utf-8")
99
+ return {
100
+ "schema_version": 1,
101
+ "arxiv_id": arxiv_id,
102
+ "bytes": len(encoded),
103
+ "sha256": hashlib.sha256(encoded).hexdigest(),
104
+ "url": url,
105
+ "converter": CONVERTER,
106
+ "converter_version": _converter_version(),
107
+ }
108
+
109
+
110
+ def _ocr_meta(arxiv_id: str, url: str, markdown: str, result) -> dict:
111
+ meta = _base_meta(arxiv_id, url, markdown)
112
+ meta.update(
113
+ {
114
+ "conversion_mode": "selective_ocr",
115
+ "quality": "partial" if result.pages_recommending_hosted else "complete",
116
+ "page_count": result.page_count,
117
+ "pages_recommended_for_ocr": list(result.pages_recommended_for_ocr),
118
+ "pages_routed_to_ocr": list(result.pages_routed_to_ocr),
119
+ "pages_recommending_hosted": list(result.pages_recommending_hosted),
120
+ "ocr_reasons_by_page": _ocr_reasons(result.ocr_reasons_by_page),
121
+ "pages_with_tables": list(result.pages_with_tables),
122
+ "pages_with_columns": list(result.pages_with_columns),
123
+ "pages": [_page_provenance(page) for page in result.pages],
124
+ }
125
+ )
126
+ return meta
127
+
128
+
129
+ def _native_fallback_meta(arxiv_id: str, url: str, markdown: str, result, error: ValueError) -> dict:
130
+ meta = _ocr_meta(arxiv_id, url, markdown, result)
131
+ meta["conversion_mode"] = "native_fallback"
132
+ meta["quality"] = "partial"
133
+ meta["ocr_error"] = str(error)
134
+ return meta
135
+
136
+
137
+ def _usable(markdown: str | None) -> bool:
138
+ return markdown is not None and len(markdown.strip()) >= MIN_MARKDOWN_CHARS
139
+
140
+
141
+ def _is_pdf(path: Path) -> bool:
142
+ try:
143
+ with path.open("rb") as stream:
144
+ return stream.read(4) == b"%PDF"
145
+ except OSError:
146
+ return False
147
+
148
+
149
+ def _download_pdf(url: str, pdf_path: Path) -> bool:
150
+ raw = _fetch(url)
151
+ if raw is None:
152
+ print(f"could not fetch {url}", file=sys.stderr)
153
+ return False
154
+ if not raw.startswith(b"%PDF"):
155
+ print(f"{url} did not return a PDF", file=sys.stderr)
156
+ return False
157
+ descriptor, staged = tempfile.mkstemp(prefix=f".{pdf_path.name}-", suffix=".part", dir=pdf_path.parent)
158
+ try:
159
+ with os.fdopen(descriptor, "wb") as stream:
160
+ stream.write(raw)
161
+ os.replace(staged, pdf_path)
162
+ finally:
163
+ try:
164
+ os.unlink(staged)
165
+ except FileNotFoundError:
166
+ pass
167
+ return True
168
+
169
+
170
+ def convert(arxiv_id: str, *, allow_native_fallback: bool = False) -> bool:
171
+ d = CACHE_DIR / arxiv_id
172
+ cached = _cached_conversion(d, allow_native_fallback=allow_native_fallback)
173
+ if cached is not None:
174
+ print(f"{arxiv_id}: cached at {cached}")
175
+ return True
176
+
177
+ try:
178
+ import pdf_inspector
179
+ except ImportError:
180
+ print(
181
+ "PDF conversion needs `paperstack-cli[pdf]`; reinstall with "
182
+ "`uv tool install --force 'paperstack-cli[pdf]'`.",
183
+ file=sys.stderr,
184
+ )
185
+ return False
186
+
187
+ url = f"https://arxiv.org/pdf/{arxiv_id}"
188
+ d.mkdir(parents=True, exist_ok=True)
189
+ pdf_path = d / "paper.pdf"
190
+ reused_cached_pdf = _is_pdf(pdf_path)
191
+ if not reused_cached_pdf and not _download_pdf(url, pdf_path):
192
+ return False
193
+
194
+ while True:
195
+ ocr_error = None
196
+ try:
197
+ result = pdf_inspector.process_pdf_with_ocr(str(pdf_path), mode="auto")
198
+ parsed = result
199
+ md = result.markdown
200
+ meta = _ocr_meta(arxiv_id, url, md, result)
201
+ except ValueError as error:
202
+ ocr_error = error
203
+ try:
204
+ native = pdf_inspector.process_pdf_with_ocr(str(pdf_path), mode="off")
205
+ except ValueError:
206
+ native = None
207
+ parsed = native
208
+ if native is None:
209
+ md = None
210
+ else:
211
+ md = native.markdown
212
+ if _usable(md):
213
+ meta = _native_fallback_meta(arxiv_id, url, md, native, error)
214
+
215
+ if _usable(md):
216
+ if ocr_error is not None:
217
+ print(f"{arxiv_id}: OCR unavailable; cached {meta['quality']} native extraction", file=sys.stderr)
218
+ break
219
+ needs_ocr = parsed is not None and bool(parsed.pages_recommended_for_ocr)
220
+ if reused_cached_pdf and not needs_ocr and _download_pdf(url, pdf_path):
221
+ reused_cached_pdf = False
222
+ continue
223
+ if ocr_error is not None:
224
+ print(f"{arxiv_id}: PDF conversion failed: {ocr_error}", file=sys.stderr)
225
+ if needs_ocr:
226
+ print(
227
+ "OCR needs PDFium and ONNX Runtime; set PDFIUM_LIB_PATH and ORT_DYLIB_PATH.",
228
+ file=sys.stderr,
229
+ )
230
+ print(f"OCR runtime setup: {OCR_RUNTIME_GUIDE}", file=sys.stderr)
231
+ return False
232
+ chars = len(md.strip()) if md else 0
233
+ print(f"{arxiv_id}: converted to only {chars} chars, treat as a failure", file=sys.stderr)
234
+ return False
235
+
236
+ md_path = d / "paper.md"
237
+ md_path.write_bytes(md.encode("utf-8"))
238
+ (d / "meta.json").write_text(json.dumps(meta, indent=2), encoding="utf-8")
239
+ print(f"{arxiv_id}: {len(md)} chars -> {md_path}")
240
+ return True
241
+
242
+
243
+ def main(argv: list[str]) -> None:
244
+ if not argv:
245
+ sys.exit("pass one or more arXiv ids")
246
+ CACHE_DIR.mkdir(parents=True, exist_ok=True)
247
+ failures = [a for a in argv if not convert(a)]
248
+ if failures:
249
+ sys.exit(f"failed: {' '.join(failures)}")
250
+
251
+
252
+ if __name__ == "__main__":
253
+ main(sys.argv[1:])
@@ -9,6 +9,8 @@ from dotenv import find_dotenv, load_dotenv
9
9
 
10
10
  from . import credentials
11
11
 
12
+ _EXPORTED_ONLY = ("PDFIUM_LIB_PATH", "ORT_DYLIB_PATH")
13
+
12
14
 
13
15
  def load_environment() -> None:
14
16
  """Load the nearest .env without replacing exported variables."""
@@ -16,6 +18,9 @@ def load_environment() -> None:
16
18
  path = find_dotenv(usecwd=True)
17
19
  if path:
18
20
  load_dotenv(path, override=False)
21
+ for name in _EXPORTED_ONLY:
22
+ if name not in exported_keys:
23
+ os.environ.pop(name, None)
19
24
  credentials.set_environment_context(exported_keys, Path(path) if path else None)
20
25
 
21
26
 
@@ -6,11 +6,18 @@ import json
6
6
  import os
7
7
  import re
8
8
  import time
9
+ import unicodedata
9
10
  import urllib.error
10
11
  import urllib.parse
11
12
  import urllib.request
12
13
  import xml.etree.ElementTree as ET
14
+ from contextlib import contextmanager
13
15
  from dataclasses import dataclass
16
+ from datetime import UTC, datetime
17
+ from email.utils import parsedate_to_datetime
18
+ from pathlib import Path
19
+
20
+ from filelock import FileLock
14
21
 
15
22
  from . import credentials
16
23
 
@@ -45,6 +52,59 @@ class PaperRef:
45
52
 
46
53
 
47
54
  _last_request: dict[str, float] = {}
55
+ _OPENREVIEW_HOSTS = {"api.openreview.net", "api2.openreview.net"}
56
+
57
+
58
+ @contextmanager
59
+ def _request_slot(key: str):
60
+ if key != "openreview":
61
+ yield None
62
+ return
63
+ base = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
64
+ root = base / "paperstack" / "http"
65
+ try:
66
+ root.mkdir(parents=True, exist_ok=True, mode=0o700)
67
+ lock = FileLock(root / "openreview.lock", timeout=120)
68
+ lock.acquire()
69
+ except OSError:
70
+ yield None
71
+ return
72
+ try:
73
+ yield root / "openreview.timestamp"
74
+ finally:
75
+ lock.release()
76
+
77
+
78
+ def _shared_elapsed(path: Path | None) -> float:
79
+ if path is None:
80
+ return float("inf")
81
+ try:
82
+ return max(0.0, time.time() - path.stat().st_mtime)
83
+ except OSError:
84
+ return float("inf")
85
+
86
+
87
+ def _mark_request(path: Path | None) -> None:
88
+ if path is not None:
89
+ try:
90
+ path.touch()
91
+ except OSError:
92
+ pass
93
+
94
+
95
+ def _retry_delay(exc: urllib.error.HTTPError, attempt: int) -> float:
96
+ value = exc.headers.get("Retry-After") if exc.headers else None
97
+ if value:
98
+ try:
99
+ delay = float(value)
100
+ except ValueError:
101
+ try:
102
+ delay = (parsedate_to_datetime(value) - datetime.now(UTC)).total_seconds()
103
+ except (TypeError, ValueError, OverflowError):
104
+ delay = -1
105
+ if delay >= 0:
106
+ return min(delay, 120)
107
+ return min(2**attempt, 30)
48
108
 
49
109
 
50
110
  def request(
@@ -56,34 +116,41 @@ def request(
56
116
  if params:
57
117
  url += "?" + urllib.parse.urlencode(params)
58
118
  host = urllib.parse.urlparse(url).netloc
119
+ request_key = "openreview" if host in _OPENREVIEW_HOSTS else host
59
120
  interval = 3.0 if "arxiv.org" in host else 1.1 if "dblp.org" in host else 0.5
60
121
  request_headers = {
61
122
  "User-Agent": "paperstack (+https://github.com/MilkClouds/paperstack)",
62
123
  **(headers or {}),
63
124
  }
64
125
  req = urllib.request.Request(url, headers=request_headers, data=data)
65
- for attempt in range(3):
66
- elapsed = time.monotonic() - _last_request.get(host, 0.0)
67
- if elapsed < interval:
68
- time.sleep(interval - elapsed)
69
- try:
70
- with urllib.request.urlopen(req, timeout=30) as response:
71
- _last_request[host] = time.monotonic()
72
- return response.read()
73
- except urllib.error.HTTPError as exc:
74
- _last_request[host] = time.monotonic()
75
- if exc.code != 429:
76
- raise
77
- if attempt == 2:
78
- has_api_key = any(name.lower() == "x-api-key" for name in request_headers)
79
- if host == "api.semanticscholar.org" and not has_api_key:
80
- message = (
81
- f"{exc.reason}; configure semantic-scholar.api-key with "
82
- "`paperstack config set semantic-scholar.api-key` for more reliable access"
83
- )
84
- raise urllib.error.HTTPError(exc.url, exc.code, message, exc.headers, exc.fp) from exc
85
- raise
86
- time.sleep(5 * (attempt + 1))
126
+ with _request_slot(request_key) as shared_timestamp:
127
+ for attempt in range(3):
128
+ elapsed = min(
129
+ time.monotonic() - _last_request.get(request_key, 0.0),
130
+ _shared_elapsed(shared_timestamp),
131
+ )
132
+ if elapsed < interval:
133
+ time.sleep(interval - elapsed)
134
+ try:
135
+ with urllib.request.urlopen(req, timeout=30) as response:
136
+ _last_request[request_key] = time.monotonic()
137
+ _mark_request(shared_timestamp)
138
+ return response.read()
139
+ except urllib.error.HTTPError as exc:
140
+ _last_request[request_key] = time.monotonic()
141
+ _mark_request(shared_timestamp)
142
+ if exc.code != 429:
143
+ raise
144
+ if attempt == 2:
145
+ has_api_key = any(name.lower() == "x-api-key" for name in request_headers)
146
+ if host == "api.semanticscholar.org" and not has_api_key:
147
+ message = (
148
+ f"{exc.reason}; configure semantic-scholar.api-key with "
149
+ "`paperstack config set semantic-scholar.api-key` for more reliable access"
150
+ )
151
+ raise urllib.error.HTTPError(exc.url, exc.code, message, exc.headers, exc.fp) from exc
152
+ raise
153
+ time.sleep(_retry_delay(exc, attempt))
87
154
  raise RuntimeError("unreachable request retry state")
88
155
 
89
156
 
@@ -305,11 +372,50 @@ def fetch_all(
305
372
  return results
306
373
 
307
374
 
308
- def search(source: str, query: str, *, local_only: bool = False) -> dict:
375
+ def _content_value(note: dict, name: str):
376
+ value = (note.get("content") or {}).get(name)
377
+ return value.get("value") if isinstance(value, dict) and "value" in value else value
378
+
379
+
380
+ def _normalized_title(value: object) -> str:
381
+ return re.sub(r"\W+", "", unicodedata.normalize("NFKC", str(value or "")).casefold())
382
+
383
+
384
+ def _openreview_status(note: dict) -> str:
385
+ venue = _content_value(note, "venue") or _content_value(note, "venueid") or _content_value(note, "venue_id")
386
+ fields = [*note.get("invitations", []), venue, _content_value(note, "decision"), _content_value(note, "status")]
387
+ text = " ".join(str(value) for value in fields if value).casefold()
388
+ if "withdraw" in text:
389
+ return "withdrawn"
390
+ if "reject" in text:
391
+ return "rejected"
392
+ if "accept" in text or venue and not any(word in str(venue).casefold() for word in ("submission", "submitted")):
393
+ return "accepted"
394
+ return "submission"
395
+
396
+
397
+ def search(
398
+ source: str,
399
+ query: str,
400
+ *,
401
+ limit: int = 10,
402
+ local_only: bool = False,
403
+ exact_title: bool = False,
404
+ openreview_status: str | None = None,
405
+ ) -> dict:
406
+ query = query.strip()
407
+ if not query:
408
+ raise ValueError("paper search query is required")
409
+ if not 1 <= limit <= 100:
410
+ raise ValueError("limit must be between 1 and 100")
411
+ if (exact_title or openreview_status) and source != "openreview":
412
+ raise ValueError("exact title and OpenReview status filters require the openreview source")
413
+ if openreview_status not in (None, "submission", "accepted", "withdrawn"):
414
+ raise ValueError("OpenReview status must be submission, accepted, or withdrawn")
309
415
  if source == "dblp":
310
416
  from . import dblp_index
311
417
 
312
- if hits := dblp_index.search(query):
418
+ if hits := dblp_index.search(query, limit=limit):
313
419
  return _result("dblp", str(dblp_index.index_path()), {"query": query, "matches": hits})
314
420
  if local_only:
315
421
  return {
@@ -319,19 +425,25 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
319
425
  "reason": "not found in local index",
320
426
  }
321
427
  url = "https://dblp.org/search/publ/api"
428
+ has_field_token = re.search(r"(?:^|\s)(?:author|title|venue|year|type|stream|toc):[^:]+:", query)
429
+ remote_query = query if has_field_token else re.sub(r":\s+", " ", query)
322
430
  return _safe(
323
- lambda: _result("dblp", url, _get_json(url, {"q": query, "format": "json", "h": 10})), "dblp", url
431
+ lambda: _result("dblp", url, _get_json(url, {"q": remote_query, "format": "json", "h": limit})),
432
+ "dblp",
433
+ url,
324
434
  )
325
435
  if source == "crossref":
326
436
  url = "https://api.crossref.org/works"
327
437
  return _safe(
328
- lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows": 10})), "crossref", url
438
+ lambda: _result("crossref", url, _get_json(url, {"query.title": query, "rows": limit})),
439
+ "crossref",
440
+ url,
329
441
  )
330
442
  if source == "arxiv":
331
443
  url = "https://export.arxiv.org/api/query"
332
444
 
333
445
  def arxiv_search() -> dict:
334
- root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results": 10}))
446
+ root = ET.fromstring(_get_text(url, {"search_query": f'ti:"{query}"', "max_results": limit}))
335
447
  matches = [
336
448
  {
337
449
  "id": entry.findtext("atom:id", "", ARXIV_NS),
@@ -347,40 +459,75 @@ def search(source: str, query: str, *, local_only: bool = False) -> dict:
347
459
  token = os.environ.get("OPENREVIEW_ACCESS_TOKEN")
348
460
  headers = {"Cookie": f"openreview.accessToken={token}"} if token else {}
349
461
  endpoints = (
350
- "https://api2.openreview.net/notes/search",
351
- "https://api.openreview.net/notes/search",
462
+ ("https://api2.openreview.net/notes/search", True),
463
+ ("https://api.openreview.net/notes/search", False),
352
464
  )
353
465
  matches = []
354
466
  errors = []
355
- for endpoint in endpoints:
356
- result = _safe(
357
- lambda endpoint=endpoint: _result(
467
+ for endpoint, is_v2 in endpoints:
468
+ endpoint_matches = []
469
+ local_filter = bool(openreview_status or (exact_title and not is_v2))
470
+ page_size = min(max(limit * 2, 20), 100) if local_filter else limit
471
+ params = {"limit": page_size, "source": "forum", "cache": "true"}
472
+ if exact_title and is_v2:
473
+ params.update({"term": query, "type": "exact", "content": "title"})
474
+ else:
475
+ params["query"] = query
476
+ for offset in range(0, 1000, page_size):
477
+ page_params = {**params, "offset": offset} if offset else params
478
+ result = _safe(
479
+ lambda endpoint=endpoint, page_params=page_params: _result(
480
+ "openreview",
481
+ endpoint,
482
+ _get_json(endpoint, page_params, headers),
483
+ ),
358
484
  "openreview",
359
485
  endpoint,
360
- _get_json(endpoint, {"query": query, "limit": 10, "source": "forum"}, headers),
361
- ),
362
- "openreview",
363
- endpoint,
364
- )
365
- if result["status"] == "ok":
366
- matches.extend(result.get("response", {}).get("notes", []))
367
- else:
368
- errors.append(result.get("error", endpoint))
486
+ )
487
+ if result["status"] != "ok":
488
+ errors.append(result.get("error", endpoint))
489
+ break
490
+ notes = result.get("response", {}).get("notes", [])
491
+ matches.extend(notes)
492
+ endpoint_matches.extend(notes)
493
+ if not local_filter or len(notes) < page_size:
494
+ break
495
+ eligible = endpoint_matches
496
+ if exact_title:
497
+ wanted = _normalized_title(query)
498
+ eligible = [
499
+ note for note in eligible if _normalized_title(_content_value(note, "title")) == wanted
500
+ ]
501
+ if openreview_status:
502
+ eligible = [note for note in eligible if _openreview_status(note) == openreview_status]
503
+ eligible_ids = {note.get("forum") or note.get("id") for note in eligible}
504
+ if len(eligible_ids) >= limit:
505
+ break
369
506
  if not matches and errors:
370
- return _result("openreview", endpoints[0], error="; ".join(errors))
507
+ return _result("openreview", endpoints[0][0], error="; ".join(errors))
371
508
  unique = {}
372
509
  for note in matches:
373
510
  unique[note.get("forum") or note.get("id") or json.dumps(note, sort_keys=True)] = note
511
+ selected = list(unique.values())
512
+ if exact_title:
513
+ wanted = _normalized_title(query)
514
+ selected = [note for note in selected if _normalized_title(_content_value(note, "title")) == wanted]
515
+ if openreview_status:
516
+ selected = [note for note in selected if _openreview_status(note) == openreview_status]
374
517
  return _result(
375
518
  "openreview",
376
- endpoints[0],
377
- {"query": query, "matches": list(unique.values()), "api_endpoints": list(endpoints)},
519
+ endpoints[0][0],
520
+ {
521
+ "query": query,
522
+ "matches": selected[:limit],
523
+ "api_endpoints": [endpoint for endpoint, _ in endpoints],
524
+ },
378
525
  )
379
526
  if source == "s2":
380
527
  url = "https://api.semanticscholar.org/graph/v1/paper/search"
381
528
  api_key = credentials.get(credentials.SEMANTIC_SCHOLAR_API_KEY)
382
529
  headers = {"x-api-key": api_key} if api_key else {}
383
- params = {"query": query, "limit": 10, "fields": S2_FIELDS}
530
+ params = {"query": query, "limit": limit, "fields": S2_FIELDS}
384
531
  return _safe(
385
532
  lambda: _result("semantic_scholar", url, _get_json(url, params, headers)), "semantic_scholar", url
386
533
  )
@@ -94,11 +94,12 @@ def build(root: Path, output: Path) -> int:
94
94
  ):
95
95
  raise ValueError("viewer output would replace a broad path, the corpus, or authored entries")
96
96
  backup = destination.parent / f".{destination.name}.backup"
97
- if backup.exists() and not destination.exists():
97
+ if backup.exists():
98
98
  marker = backup / ".paperstack-viewer"
99
99
  if not backup.is_dir() or not marker.is_file() or marker.read_text(encoding="utf-8") != "1\n":
100
100
  raise ValueError(f"viewer backup is not owned by Paperstack: {backup}")
101
- os.replace(backup, destination)
101
+ if not destination.exists():
102
+ os.replace(backup, destination)
102
103
  if destination.exists() and not destination.is_dir():
103
104
  raise ValueError(f"viewer output is not a directory: {destination}")
104
105
  if destination.exists() and any(destination.iterdir()):
@@ -1,103 +0,0 @@
1
- """Convert native-PDF arXiv submissions to Markdown.
2
-
3
- Used by `paperstack paper pdf` after the source fetcher reports no TeX.
4
- """
5
-
6
- from __future__ import annotations
7
-
8
- import hashlib
9
- import json
10
- import os
11
- import sys
12
- import time
13
- import urllib.error
14
- import urllib.request
15
- from pathlib import Path
16
-
17
- _CACHE_ROOT = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
18
- CACHE_DIR = Path(os.environ.get("PAPERSTACK_PAPERS_DIR", _CACHE_ROOT / "paperstack" / "papers"))
19
-
20
-
21
- def _fetch(url: str, timeout: int = 60) -> bytes | None:
22
- req = urllib.request.Request(url, headers={"User-Agent": "paperstack/1.0 (+arxiv pdf fetch)"})
23
- for attempt in range(3):
24
- try:
25
- with urllib.request.urlopen(req, timeout=timeout) as resp:
26
- return resp.read()
27
- except urllib.error.HTTPError as e:
28
- if e.code == 429:
29
- time.sleep(15 * (attempt + 1))
30
- continue
31
- if attempt == 2:
32
- return None
33
- time.sleep(5)
34
- except (urllib.error.URLError, OSError, TimeoutError):
35
- if attempt == 2:
36
- return None
37
- time.sleep(5)
38
- return None
39
-
40
-
41
- def convert(arxiv_id: str) -> bool:
42
- d = CACHE_DIR / arxiv_id
43
- md_path = d / "paper.md"
44
- if md_path.exists() and md_path.stat().st_size > 1000:
45
- print(f"{arxiv_id}: cached at {md_path}")
46
- return True
47
-
48
- try:
49
- import pymupdf4llm
50
- except ImportError:
51
- print(
52
- "PDF conversion needs `paperstack-cli[pdf]`; reinstall with "
53
- "`uv tool install --force 'paperstack-cli[pdf]'`.",
54
- file=sys.stderr,
55
- )
56
- return False
57
-
58
- url = f"https://arxiv.org/pdf/{arxiv_id}"
59
- raw = _fetch(url)
60
- if raw is None:
61
- print(f"{arxiv_id}: could not fetch {url}", file=sys.stderr)
62
- return False
63
- if not raw.startswith(b"%PDF"):
64
- print(f"{arxiv_id}: {url} did not return a PDF", file=sys.stderr)
65
- return False
66
-
67
- d.mkdir(parents=True, exist_ok=True)
68
- pdf_path = d / "paper.pdf"
69
- pdf_path.write_bytes(raw)
70
- md = pymupdf4llm.to_markdown(str(pdf_path))
71
- if len(md) < 500:
72
- print(f"{arxiv_id}: converted to only {len(md)} chars, treat as a failure", file=sys.stderr)
73
- return False
74
-
75
- md_path.write_text(md, encoding="utf-8")
76
- (d / "meta.json").write_text(
77
- json.dumps(
78
- {
79
- "arxiv_id": arxiv_id,
80
- "bytes": len(md),
81
- "sha256": hashlib.sha256(md.encode("utf-8")).hexdigest(),
82
- "url": url,
83
- "converter": "pymupdf4llm",
84
- },
85
- indent=2,
86
- ),
87
- encoding="utf-8",
88
- )
89
- print(f"{arxiv_id}: {len(md)} chars -> {md_path}")
90
- return True
91
-
92
-
93
- def main(argv: list[str]) -> None:
94
- if not argv:
95
- sys.exit("pass one or more arXiv ids")
96
- CACHE_DIR.mkdir(parents=True, exist_ok=True)
97
- failures = [a for a in argv if not convert(a)]
98
- if failures:
99
- sys.exit(f"failed: {' '.join(failures)}")
100
-
101
-
102
- if __name__ == "__main__":
103
- main(sys.argv[1:])
File without changes