bibcite-cli 0.6.3__tar.gz → 0.6.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/.github/workflows/publish.yml +12 -4
  2. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/PKG-INFO +5 -3
  3. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/Readme.md +4 -2
  4. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/pyproject.toml +1 -1
  5. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/normalize.py +1 -2
  6. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/sources.py +62 -62
  7. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/venues.py +1 -1
  8. bibcite_cli-0.6.4/tests/test_no_google_scholar.py +69 -0
  9. bibcite_cli-0.6.4/tests/test_public_service.py +188 -0
  10. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_status_semantics.py +3 -3
  11. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/uv.lock +1 -1
  12. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/.github/workflows/ci.yml +0 -0
  13. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/.gitignore +0 -0
  14. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/LICENSE +0 -0
  15. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/assets/bibcite.svg +0 -0
  16. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/skills/bibcite/SKILL.md +0 -0
  17. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/__init__.py +0 -0
  18. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/bibfile.py +0 -0
  19. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/cache.py +0 -0
  20. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/cli.py +0 -0
  21. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/data/strings.bib +0 -0
  22. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/resolve.py +0 -0
  23. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_bibfile.py +0 -0
  24. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_bugfixes.py +0 -0
  25. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_cli_status.py +0 -0
  26. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_entry_types.py +0 -0
  27. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_normalize.py +0 -0
  28. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_round2.py +0 -0
  29. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_round3.py +0 -0
  30. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_source_retries.py +0 -0
  31. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_strings_override.py +0 -0
  32. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_venues.py +0 -0
  33. {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_webpages.py +0 -0
@@ -4,6 +4,12 @@ on:
4
4
  release:
5
5
  types:
6
6
  - published
7
+ workflow_dispatch:
8
+ inputs:
9
+ tag:
10
+ description: Release tag to publish
11
+ required: true
12
+ type: string
7
13
 
8
14
  permissions:
9
15
  contents: read
@@ -12,11 +18,13 @@ jobs:
12
18
  build:
13
19
  name: Build and verify distributions
14
20
  runs-on: ubuntu-latest
21
+ env:
22
+ RELEASE_TAG: ${{ github.event.release.tag_name || inputs.tag }}
15
23
  steps:
16
24
  - name: Check out the release
17
25
  uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
18
26
  with:
19
- ref: ${{ github.event.release.tag_name }}
27
+ ref: ${{ env.RELEASE_TAG }}
20
28
 
21
29
  - name: Install uv and Python
22
30
  uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
@@ -26,8 +34,8 @@ jobs:
26
34
  - name: Verify the release tag matches the package version
27
35
  run: |
28
36
  package_version="$(uv version --short)"
29
- if [ "${GITHUB_REF_NAME}" != "v${package_version}" ]; then
30
- echo "Release tag ${GITHUB_REF_NAME} does not match package version ${package_version}."
37
+ if [ "${RELEASE_TAG}" != "v${package_version}" ]; then
38
+ echo "Release tag ${RELEASE_TAG} does not match package version ${package_version}."
31
39
  exit 1
32
40
  fi
33
41
 
@@ -71,4 +79,4 @@ jobs:
71
79
  path: dist/
72
80
 
73
81
  - name: Publish distributions to PyPI
74
- uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
82
+ uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: bibcite-cli
3
- Version: 0.6.3
3
+ Version: 0.6.4
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -158,7 +158,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
158
158
 
159
159
  ## How resolution works
160
160
 
161
- For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
161
+ For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
162
+ Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
162
163
  A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
163
164
 
164
165
  Successful published matches are cached at `~/.cache/bibcite/published.json`.
@@ -176,6 +177,7 @@ These optional environment variables improve source reliability:
176
177
  | `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
177
178
  | `S2_API_KEY` | Uses a private Semantic Scholar quota. |
178
179
  | `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
180
+ | `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
179
181
  | `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
180
182
  | `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
181
183
 
@@ -211,7 +213,7 @@ uv tool install --editable .
211
213
 
212
214
  ## Acknowledgements
213
215
 
214
- Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
216
+ Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
215
217
 
216
218
  ## License
217
219
 
@@ -145,7 +145,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
145
145
 
146
146
  ## How resolution works
147
147
 
148
- For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
148
+ For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
149
+ Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
149
150
  A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
150
151
 
151
152
  Successful published matches are cached at `~/.cache/bibcite/published.json`.
@@ -163,6 +164,7 @@ These optional environment variables improve source reliability:
163
164
  | `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
164
165
  | `S2_API_KEY` | Uses a private Semantic Scholar quota. |
165
166
  | `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
167
+ | `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
166
168
  | `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
167
169
  | `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
168
170
 
@@ -198,7 +200,7 @@ uv tool install --editable .
198
200
 
199
201
  ## Acknowledgements
200
202
 
201
- Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
203
+ Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
202
204
 
203
205
  ## License
204
206
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.6.3"
3
+ version = "0.6.4"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -30,8 +30,7 @@ def mini_hash(s: str, replace: str = "") -> str:
30
30
  """PaperMemory's miniHash: lowercase, non-alphanumeric replaced.
31
31
 
32
32
  When ``replace`` is non-empty, each non-word char maps to one replacement
33
- char so string positions are preserved (needed by the Google Scholar
34
- parser).
33
+ char so string positions are preserved.
35
34
  """
36
35
  if replace:
37
36
  return re.sub(r"[^a-z0-9_]", replace, s.lower())
@@ -1,7 +1,7 @@
1
1
  """API clients for the publication-matching cascade.
2
2
 
3
3
  Order and matching rules ported from PaperMemory's bibMatcher:
4
- DBLP -> Semantic Scholar -> Google Scholar -> CrossRef -> Unpaywall.
4
+ DBLP -> Semantic Scholar -> CrossRef -> Unpaywall -> OpenAlex.
5
5
  All matchers verify identity via normalized-title equality and reject
6
6
  preprint venues (arXiv / CoRR / bioRxiv / ...).
7
7
  """
@@ -15,6 +15,7 @@ import time
15
15
  import xml.etree.ElementTree as ET
16
16
  from concurrent.futures import ThreadPoolExecutor, as_completed
17
17
  from dataclasses import dataclass, field
18
+ from urllib.parse import urlsplit
18
19
 
19
20
  import httpx
20
21
 
@@ -79,8 +80,64 @@ def _request_timeout(cap: float = TIMEOUT) -> float:
79
80
  def _get(
80
81
  c: httpx.Client, url: str, *, timeout: float = TIMEOUT, **kwargs
81
82
  ) -> httpx.Response:
82
- """GET with the per-request cap narrowed by the active total deadline."""
83
- return c.get(url, timeout=_request_timeout(timeout), **kwargs)
83
+ """GET directly or route keyless supported APIs through the public service."""
84
+ public_url = os.environ.get("BIBCITE_PUBLIC_SERVICE_URL")
85
+ target = urlsplit(url)
86
+ providers = {
87
+ "api.openalex.org": "openalex",
88
+ "api.semanticscholar.org": "semanticscholar",
89
+ "api.crossref.org": "crossref",
90
+ }
91
+ provider = providers.get((target.hostname or "").lower())
92
+ personal_access = {
93
+ "openalex": bool(os.environ.get("OPENALEX_API_KEY")),
94
+ "semanticscholar": bool(
95
+ os.environ.get("S2_API_KEY")
96
+ or os.environ.get("SEMANTIC_SCHOLAR_API_KEY")
97
+ ),
98
+ "crossref": bool(os.environ.get("BIBCITE_MAILTO")),
99
+ }
100
+ if not public_url or not provider or personal_access[provider]:
101
+ return c.get(url, timeout=_request_timeout(timeout), **kwargs)
102
+
103
+ service = urlsplit(public_url)
104
+ loopback = service.hostname in {"localhost", "127.0.0.1", "::1"}
105
+ if (
106
+ (service.scheme != "https" and not (service.scheme == "http" and loopback))
107
+ or not service.hostname
108
+ or service.username is not None
109
+ or service.password is not None
110
+ or service.query
111
+ or service.fragment
112
+ ):
113
+ raise SourceUnavailable("BIBCITE_PUBLIC_SERVICE_URL is invalid")
114
+
115
+ params = kwargs.get("params") or {}
116
+ safe_params = {
117
+ str(key): str(value)
118
+ for key, value in params.items()
119
+ if str(key).lower() not in {"api_key", "mailto"}
120
+ }
121
+ try:
122
+ response = c.post(
123
+ public_url,
124
+ json={"provider": provider, "path": target.path, "params": safe_params},
125
+ timeout=_request_timeout(timeout),
126
+ follow_redirects=False,
127
+ )
128
+ except httpx.HTTPError as e:
129
+ raise SourceUnavailable(
130
+ f"public literature service unavailable ({type(e).__name__})"
131
+ ) from e
132
+ if response.status_code == 404:
133
+ return response
134
+ if response.status_code == 429:
135
+ raise SourceUnavailable("public literature service rate-limited (429)")
136
+ if response.is_error:
137
+ raise SourceUnavailable(
138
+ f"public literature service error ({response.status_code})"
139
+ )
140
+ return response
84
141
 
85
142
 
86
143
  def _sleep(delay: float):
@@ -539,62 +596,6 @@ def try_semantic_scholar(
539
596
  return None
540
597
 
541
598
 
542
- # ---------------------------------------------------------------------------
543
- # Google Scholar (port of PaperMemory's background fetchGSData)
544
- # ---------------------------------------------------------------------------
545
-
546
- def try_google_scholar(title: str) -> Match | None:
547
- with _client(browser=True) as c:
548
- r = _get(
549
- c,
550
- "https://scholar.google.com/scholar",
551
- params={"q": title, "hl": "en"},
552
- )
553
- if r.status_code == 429 or "captcha" in r.text.lower()[:5000]:
554
- raise SourceUnavailable("Google Scholar is blocking requests (captcha/429)")
555
- r.raise_for_status()
556
- parts = r.text.split("gs_res_ccl_mid")
557
- if len(parts) < 2:
558
- return None
559
- page = parts[1]
560
- # Each result title anchor looks like <a id="DATAID" href=...>Title</a>
561
- # (the title may contain <b> highlights and HTML entities).
562
- data_id = ""
563
- for am in re.finditer(
564
- r'<a[^>]*\bid="([\w-]{6,40})"[^>]*>(.*?)</a>', page, re.S
565
- ):
566
- text = html.unescape(re.sub(r"<[^>]+>", "", am.group(2)))
567
- if norm_title(text) == norm_title(title):
568
- data_id = am.group(1)
569
- break
570
- if not data_id:
571
- return None
572
- cite_url = (
573
- "https://scholar.google.com/scholar?q=info:"
574
- f"{data_id}:scholar.google.com/&output=cite&scirp=0&hl=en"
575
- )
576
- cite_html = _get(c, cite_url).text
577
- bm = re.search(r'<a[^>]*href="([^">]+)"[^>]*>BibTex</a>', cite_html, re.I)
578
- if not bm:
579
- return None
580
- bib_url = re.sub(r"\s+", "", bm.group(1).replace("&amp;", "&"))
581
- bibtex = _get(c, bib_url).text
582
- from .bibfile import parse_bibtex_entry # local import to avoid cycle
583
-
584
- entry = parse_bibtex_entry(bibtex)
585
- venue = entry.get("journal", "") or entry.get("booktitle", "")
586
- if venue and not venue.lower().endswith("xiv") and "preprint" not in venue.lower():
587
- _log(f"[googlescholar] match: {venue}")
588
- return Match(
589
- source="googlescholar",
590
- venue=venue,
591
- title=clean_title(entry.get("title", title)),
592
- year=entry.get("year", ""),
593
- bibtex=bibtex,
594
- )
595
- return None
596
-
597
-
598
599
  # ---------------------------------------------------------------------------
599
600
  # CrossRef
600
601
  # ---------------------------------------------------------------------------
@@ -827,7 +828,6 @@ def crossref_by_doi(doi: str) -> Match | None:
827
828
  CASCADE = (
828
829
  ("dblp", lambda t, y, a, au: try_dblp(t, au)),
829
830
  ("semanticscholar", lambda t, y, a, au: try_semantic_scholar(t, y, a)),
830
- ("googlescholar", lambda t, y, a, au: try_google_scholar(t)),
831
831
  ("crossref", lambda t, y, a, au: try_crossref(t)),
832
832
  ("unpaywall", lambda t, y, a, au: try_unpaywall(t)),
833
833
  ("openalex", lambda t, y, a, au: try_openalex(t)),
@@ -840,8 +840,8 @@ CASCADE = (
840
840
  _DISABLED: dict[str, str] = {}
841
841
 
842
842
  # Only these sources are authoritative enough that losing one taints a miss
843
- # into "incomplete". Google Scholar captchas and Unpaywall flakiness are
844
- # routine and must not stop "not_found" from ever being trustworthy.
843
+ # into "incomplete". Unpaywall flakiness is routine and must not stop
844
+ # "not_found" from ever being trustworthy.
845
845
  # Override with BIBCITE_CORE_SOURCES="dblp,semanticscholar" if one of these
846
846
  # is down for days and keeps every verdict incomplete.
847
847
  CORE_SOURCES = frozenset(
@@ -2,7 +2,7 @@
2
2
 
3
3
  Parses the vendored ``data/strings.bib`` @string table (journals /
4
4
  conferences / workshops) and maps venue strings returned by DBLP, Semantic
5
- Scholar, Google Scholar, CrossRef, Unpaywall, etc. onto the canonical names.
5
+ Scholar, CrossRef, Unpaywall, etc. onto the canonical names.
6
6
  """
7
7
 
8
8
  import re
@@ -0,0 +1,69 @@
1
+ """Exercise the CLI publication cascade without contacting external services."""
2
+
3
+ import importlib
4
+ import json
5
+
6
+ import httpx
7
+ import pytest
8
+
9
+ from bibcite import cache, cli, sources
10
+
11
+ resolver = importlib.import_module("bibcite.resolve")
12
+
13
+
14
+ @pytest.mark.parametrize("operation", ["get", "add", "upgrade"])
15
+ def test_cli_never_queries_google_scholar(operation, monkeypatch, tmp_path, capsys):
16
+ monkeypatch.setattr(cache, "DISABLED", True)
17
+ monkeypatch.setattr(sources, "_DISABLED", {})
18
+ monkeypatch.setattr(
19
+ resolver,
20
+ "arxiv_metadata",
21
+ lambda _: sources.ArxivMeta(
22
+ "1706.03762",
23
+ "Attention Is All You Need",
24
+ ["Ashish Vaswani"],
25
+ "2017",
26
+ "https://arxiv.org/abs/1706.03762",
27
+ ),
28
+ )
29
+ visited = []
30
+ for name in ("dblp", "semantic_scholar", "crossref", "unpaywall", "openalex"):
31
+
32
+ def miss(*args, source=name):
33
+ visited.append(source)
34
+ return None
35
+
36
+ monkeypatch.setattr(sources, f"try_{name}", miss)
37
+ monkeypatch.setattr(sources, "try_dblp_fuzzy", lambda *args: None)
38
+ requests = []
39
+
40
+ def unexpected_request(client, url, **kwargs):
41
+ requests.append(url)
42
+ return httpx.Response(429, request=httpx.Request("GET", url))
43
+
44
+ monkeypatch.setattr(sources, "_get", unexpected_request)
45
+ path = tmp_path / "references.bib"
46
+ if operation == "get":
47
+ args = ["get", "--json", "1706.03762"]
48
+ elif operation == "add":
49
+ args = ["add", "--no-tidy", str(path), "1706.03762"]
50
+ else:
51
+ path.write_text("""@article{vaswani2017attention,
52
+ title = {Attention Is All You Need},
53
+ author = {Ashish Vaswani},
54
+ year = {2017},
55
+ journal = {arXiv preprint arXiv:1706.03762},
56
+ eprint = {1706.03762}
57
+ }
58
+ """)
59
+ args = ["upgrade", str(path), "--dry-run"]
60
+ cli.main(args)
61
+ json.loads(capsys.readouterr().out)
62
+ assert set(visited) == {
63
+ "dblp",
64
+ "semantic_scholar",
65
+ "crossref",
66
+ "unpaywall",
67
+ "openalex",
68
+ }
69
+ assert requests == []
@@ -0,0 +1,188 @@
1
+ import json
2
+
3
+ import httpx
4
+ import pytest
5
+
6
+ from bibcite import sources
7
+
8
+
9
+ @pytest.fixture(autouse=True)
10
+ def clean_environment(monkeypatch):
11
+ for name in (
12
+ "BIBCITE_PUBLIC_SERVICE_URL",
13
+ "BIBCITE_MAILTO",
14
+ "OPENALEX_API_KEY",
15
+ "S2_API_KEY",
16
+ "SEMANTIC_SCHOLAR_API_KEY",
17
+ ):
18
+ monkeypatch.delenv(name, raising=False)
19
+ monkeypatch.setattr(sources, "_LAST_REQUEST", {})
20
+ monkeypatch.setattr(sources.time, "sleep", lambda _: None)
21
+
22
+
23
+ def _client(handler):
24
+ return httpx.Client(transport=httpx.MockTransport(handler))
25
+
26
+
27
+ @pytest.mark.parametrize(
28
+ ("url", "provider", "path"),
29
+ (
30
+ ("https://api.openalex.org/works", "openalex", "/works"),
31
+ (
32
+ "https://api.semanticscholar.org/graph/v1/paper/search",
33
+ "semanticscholar",
34
+ "/graph/v1/paper/search",
35
+ ),
36
+ (
37
+ "https://api.crossref.org/works/10.1234/example/transform/application/x-bibtex",
38
+ "crossref",
39
+ "/works/10.1234/example/transform/application/x-bibtex",
40
+ ),
41
+ ),
42
+ )
43
+ def test_keyless_supported_requests_route_to_public_service(
44
+ url, provider, path, monkeypatch
45
+ ):
46
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
47
+ requests = []
48
+
49
+ def handler(request):
50
+ requests.append(request)
51
+ return httpx.Response(200, json={"unchanged": True})
52
+
53
+ with _client(handler) as client:
54
+ response = sources._get(
55
+ client,
56
+ url,
57
+ params={
58
+ "query": "paper",
59
+ "limit": 5,
60
+ "api_key": "secret",
61
+ "mailto": "secret@example.com",
62
+ },
63
+ headers={"x-api-key": "secret"},
64
+ )
65
+
66
+ assert response.json() == {"unchanged": True}
67
+ assert len(requests) == 1
68
+ request = requests[0]
69
+ assert request.method == "POST"
70
+ assert request.url == "https://literature.test/v1/query"
71
+ assert "x-api-key" not in request.headers
72
+ payload = json.loads(request.content)
73
+ assert payload == {
74
+ "provider": provider,
75
+ "path": path,
76
+ "params": {"query": "paper", "limit": "5"},
77
+ }
78
+ assert "secret" not in request.content.decode()
79
+
80
+
81
+ @pytest.mark.parametrize(
82
+ ("variable", "url", "params"),
83
+ (
84
+ ("OPENALEX_API_KEY", "https://api.openalex.org/works", {"api_key": "mine"}),
85
+ ("S2_API_KEY", "https://api.semanticscholar.org/graph/v1/paper/search", {}),
86
+ (
87
+ "BIBCITE_MAILTO",
88
+ "https://api.crossref.org/works",
89
+ {"mailto": "me@example.com"},
90
+ ),
91
+ ),
92
+ )
93
+ def test_personal_access_overrides_public_service(variable, url, params, monkeypatch):
94
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
95
+ monkeypatch.setenv(variable, "mine")
96
+ requests = []
97
+
98
+ def handler(request):
99
+ requests.append(request)
100
+ return httpx.Response(200, json={})
101
+
102
+ with _client(handler) as client:
103
+ sources._get(client, url, params=params)
104
+
105
+ assert len(requests) == 1
106
+ assert requests[0].method == "GET"
107
+ assert requests[0].url.host != "literature.test"
108
+
109
+
110
+ def test_without_public_service_requests_remain_direct():
111
+ requests = []
112
+
113
+ def handler(request):
114
+ requests.append(request)
115
+ return httpx.Response(200, json={})
116
+
117
+ with _client(handler) as client:
118
+ sources._get(client, "https://api.openalex.org/works", params={"search": "x"})
119
+
120
+ assert len(requests) == 1
121
+ assert requests[0].method == "GET"
122
+ assert requests[0].url.host == "api.openalex.org"
123
+
124
+
125
+ @pytest.mark.parametrize("status", [429, 500])
126
+ def test_public_service_http_failure_is_not_retried(status, monkeypatch):
127
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
128
+ calls = 0
129
+
130
+ def handler(request):
131
+ nonlocal calls
132
+ calls += 1
133
+ return httpx.Response(status)
134
+
135
+ with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
136
+ sources._paced_get(
137
+ client,
138
+ "https://api.semanticscholar.org/graph/v1/paper/search",
139
+ "semanticscholar",
140
+ 0,
141
+ )
142
+
143
+ assert calls == 1
144
+
145
+
146
+ def test_public_service_timeout_is_not_retried(monkeypatch):
147
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
148
+ calls = 0
149
+
150
+ def handler(request):
151
+ nonlocal calls
152
+ calls += 1
153
+ raise httpx.ReadTimeout("timed out", request=request)
154
+
155
+ with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
156
+ sources._paced_get(
157
+ client,
158
+ "https://api.semanticscholar.org/graph/v1/paper/search",
159
+ "semanticscholar",
160
+ 0,
161
+ )
162
+
163
+ assert calls == 1
164
+
165
+
166
+ def test_public_service_404_is_preserved(monkeypatch):
167
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
168
+
169
+ with _client(lambda request: httpx.Response(404)) as client:
170
+ response = sources._get(client, "https://api.openalex.org/works/W1")
171
+
172
+ assert response.status_code == 404
173
+
174
+
175
+ @pytest.mark.parametrize(
176
+ "value",
177
+ (
178
+ "http://literature.test/v1/query",
179
+ "https://user:pass@literature.test/v1/query",
180
+ "https://literature.test/v1/query?secret=x",
181
+ "https://literature.test/v1/query#fragment",
182
+ ),
183
+ )
184
+ def test_invalid_public_service_url_is_rejected(value, monkeypatch):
185
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", value)
186
+ with _client(lambda request: httpx.Response(200)) as client:
187
+ with pytest.raises(sources.SourceUnavailable):
188
+ sources._get(client, "https://api.openalex.org/works")
@@ -43,7 +43,7 @@ def test_core_source_429_taints_verdict(monkeypatch):
43
43
  monkeypatch.setattr(
44
44
  sources,
45
45
  "CASCADE",
46
- _cascade(dblp="raise", googlescholar=None, crossref=None),
46
+ _cascade(dblp="raise", unpaywall=None, crossref=None),
47
47
  )
48
48
  match, status = find_published("Some Title", author_hint="smith")
49
49
  assert (match, status) == (None, "incomplete")
@@ -60,11 +60,11 @@ def test_previously_disabled_core_source_taints_next_queries(monkeypatch):
60
60
 
61
61
 
62
62
  def test_noncore_outage_does_not_taint(monkeypatch):
63
- # Google Scholar captcha is routine; a miss stays trustworthy.
63
+ # Unpaywall outages are routine; a miss stays trustworthy.
64
64
  monkeypatch.setattr(
65
65
  sources,
66
66
  "CASCADE",
67
- _cascade(dblp=None, googlescholar="raise", crossref=None),
67
+ _cascade(dblp=None, unpaywall="raise", crossref=None),
68
68
  )
69
69
  match, status = find_published("Some Title", author_hint="smith")
70
70
  assert (match, status) == (None, "not_found")
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.6.3"
21
+ version = "0.6.4"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes