bibcite-cli 0.6.2__tar.gz → 0.6.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/.github/workflows/publish.yml +12 -4
  2. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/PKG-INFO +6 -4
  3. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/Readme.md +4 -2
  4. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/pyproject.toml +1 -1
  5. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/normalize.py +1 -2
  6. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/sources.py +159 -76
  7. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/venues.py +1 -1
  8. bibcite_cli-0.6.4/tests/test_no_google_scholar.py +69 -0
  9. bibcite_cli-0.6.4/tests/test_public_service.py +188 -0
  10. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_source_retries.py +42 -2
  11. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_status_semantics.py +3 -3
  12. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/uv.lock +1 -1
  13. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/.github/workflows/ci.yml +0 -0
  14. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/.gitignore +0 -0
  15. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/LICENSE +0 -0
  16. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/assets/bibcite.svg +0 -0
  17. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/skills/bibcite/SKILL.md +0 -0
  18. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/__init__.py +0 -0
  19. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/bibfile.py +0 -0
  20. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/cache.py +0 -0
  21. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/cli.py +0 -0
  22. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/data/strings.bib +0 -0
  23. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/resolve.py +0 -0
  24. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_bibfile.py +0 -0
  25. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_bugfixes.py +0 -0
  26. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_cli_status.py +0 -0
  27. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_entry_types.py +0 -0
  28. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_normalize.py +0 -0
  29. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_round2.py +0 -0
  30. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_round3.py +0 -0
  31. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_strings_override.py +0 -0
  32. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_venues.py +0 -0
  33. {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_webpages.py +0 -0
@@ -4,6 +4,12 @@ on:
4
4
  release:
5
5
  types:
6
6
  - published
7
+ workflow_dispatch:
8
+ inputs:
9
+ tag:
10
+ description: Release tag to publish
11
+ required: true
12
+ type: string
7
13
 
8
14
  permissions:
9
15
  contents: read
@@ -12,11 +18,13 @@ jobs:
12
18
  build:
13
19
  name: Build and verify distributions
14
20
  runs-on: ubuntu-latest
21
+ env:
22
+ RELEASE_TAG: ${{ github.event.release.tag_name || inputs.tag }}
15
23
  steps:
16
24
  - name: Check out the release
17
25
  uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
18
26
  with:
19
- ref: ${{ github.event.release.tag_name }}
27
+ ref: ${{ env.RELEASE_TAG }}
20
28
 
21
29
  - name: Install uv and Python
22
30
  uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
@@ -26,8 +34,8 @@ jobs:
26
34
  - name: Verify the release tag matches the package version
27
35
  run: |
28
36
  package_version="$(uv version --short)"
29
- if [ "${GITHUB_REF_NAME}" != "v${package_version}" ]; then
30
- echo "Release tag ${GITHUB_REF_NAME} does not match package version ${package_version}."
37
+ if [ "${RELEASE_TAG}" != "v${package_version}" ]; then
38
+ echo "Release tag ${RELEASE_TAG} does not match package version ${package_version}."
31
39
  exit 1
32
40
  fi
33
41
 
@@ -71,4 +79,4 @@ jobs:
71
79
  path: dist/
72
80
 
73
81
  - name: Publish distributions to PyPI
74
- uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
82
+ uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: bibcite-cli
3
- Version: 0.6.2
3
+ Version: 0.6.4
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -158,7 +158,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
158
158
 
159
159
  ## How resolution works
160
160
 
161
- For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
161
+ For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
162
+ Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
162
163
  A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
163
164
 
164
165
  Successful published matches are cached at `~/.cache/bibcite/published.json`.
@@ -176,6 +177,7 @@ These optional environment variables improve source reliability:
176
177
  | `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
177
178
  | `S2_API_KEY` | Uses a private Semantic Scholar quota. |
178
179
  | `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
180
+ | `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
179
181
  | `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
180
182
  | `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
181
183
 
@@ -211,7 +213,7 @@ uv tool install --editable .
211
213
 
212
214
  ## Acknowledgements
213
215
 
214
- Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
216
+ Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
215
217
 
216
218
  ## License
217
219
 
@@ -145,7 +145,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
145
145
 
146
146
  ## How resolution works
147
147
 
148
- For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
148
+ For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
149
+ Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
149
150
  A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
150
151
 
151
152
  Successful published matches are cached at `~/.cache/bibcite/published.json`.
@@ -163,6 +164,7 @@ These optional environment variables improve source reliability:
163
164
  | `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
164
165
  | `S2_API_KEY` | Uses a private Semantic Scholar quota. |
165
166
  | `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
167
+ | `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
166
168
  | `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
167
169
  | `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
168
170
 
@@ -198,7 +200,7 @@ uv tool install --editable .
198
200
 
199
201
  ## Acknowledgements
200
202
 
201
- Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
203
+ Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
202
204
 
203
205
  ## License
204
206
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.6.2"
3
+ version = "0.6.4"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -30,8 +30,7 @@ def mini_hash(s: str, replace: str = "") -> str:
30
30
  """PaperMemory's miniHash: lowercase, non-alphanumeric replaced.
31
31
 
32
32
  When ``replace`` is non-empty, each non-word char maps to one replacement
33
- char so string positions are preserved (needed by the Google Scholar
34
- parser).
33
+ char so string positions are preserved.
35
34
  """
36
35
  if replace:
37
36
  return re.sub(r"[^a-z0-9_]", replace, s.lower())
@@ -1,7 +1,7 @@
1
1
  """API clients for the publication-matching cascade.
2
2
 
3
3
  Order and matching rules ported from PaperMemory's bibMatcher:
4
- DBLP -> Semantic Scholar -> Google Scholar -> CrossRef -> Unpaywall.
4
+ DBLP -> Semantic Scholar -> CrossRef -> Unpaywall -> OpenAlex.
5
5
  All matchers verify identity via normalized-title equality and reject
6
6
  preprint venues (arXiv / CoRR / bioRxiv / ...).
7
7
  """
@@ -10,10 +10,12 @@ import html
10
10
  import os
11
11
  import re
12
12
  import sys
13
+ import threading
13
14
  import time
14
15
  import xml.etree.ElementTree as ET
15
16
  from concurrent.futures import ThreadPoolExecutor, as_completed
16
17
  from dataclasses import dataclass, field
18
+ from urllib.parse import urlsplit
17
19
 
18
20
  import httpx
19
21
 
@@ -30,6 +32,10 @@ BROWSER_UA = (
30
32
  # never a false "not published". The arXiv metadata fetch sets its own longer
31
33
  # timeout on the request itself, so this does not affect it.
32
34
  TIMEOUT = 8.0
35
+ # Publication matching is enrichment on top of a valid arXiv citation. Keep
36
+ # the entire concurrent cascade, including its DBLP title-drift fallback,
37
+ # within an interactive budget instead of letting sequential retries add up.
38
+ PUBLICATION_TIMEOUT = 10.0
33
39
 
34
40
  PREPRINT_VENUES = re.compile(r"arxiv|corr|biorxiv|medrxiv|chemrxiv|ssrn|preprint", re.I)
35
41
  ARXIV_DOI = re.compile(r"^10\.48550/", re.I)
@@ -54,6 +60,112 @@ class TransientSourceError(SourceUnavailable):
54
60
  process-wide circuit breaker for later batch entries."""
55
61
 
56
62
 
63
+ class PublicationTimeout(TransientSourceError):
64
+ """The total publication-matching budget was exhausted."""
65
+
66
+
67
+ _REQUEST_DEADLINE = threading.local()
68
+
69
+
70
+ def _request_timeout(cap: float = TIMEOUT) -> float:
71
+ deadline = getattr(_REQUEST_DEADLINE, "value", None)
72
+ if deadline is None:
73
+ return cap
74
+ remaining = deadline - time.monotonic()
75
+ if remaining <= 0:
76
+ raise PublicationTimeout("publication lookup timed out")
77
+ return max(0.001, min(cap, remaining))
78
+
79
+
80
+ def _get(
81
+ c: httpx.Client, url: str, *, timeout: float = TIMEOUT, **kwargs
82
+ ) -> httpx.Response:
83
+ """GET directly or route keyless supported APIs through the public service."""
84
+ public_url = os.environ.get("BIBCITE_PUBLIC_SERVICE_URL")
85
+ target = urlsplit(url)
86
+ providers = {
87
+ "api.openalex.org": "openalex",
88
+ "api.semanticscholar.org": "semanticscholar",
89
+ "api.crossref.org": "crossref",
90
+ }
91
+ provider = providers.get((target.hostname or "").lower())
92
+ personal_access = {
93
+ "openalex": bool(os.environ.get("OPENALEX_API_KEY")),
94
+ "semanticscholar": bool(
95
+ os.environ.get("S2_API_KEY")
96
+ or os.environ.get("SEMANTIC_SCHOLAR_API_KEY")
97
+ ),
98
+ "crossref": bool(os.environ.get("BIBCITE_MAILTO")),
99
+ }
100
+ if not public_url or not provider or personal_access[provider]:
101
+ return c.get(url, timeout=_request_timeout(timeout), **kwargs)
102
+
103
+ service = urlsplit(public_url)
104
+ loopback = service.hostname in {"localhost", "127.0.0.1", "::1"}
105
+ if (
106
+ (service.scheme != "https" and not (service.scheme == "http" and loopback))
107
+ or not service.hostname
108
+ or service.username is not None
109
+ or service.password is not None
110
+ or service.query
111
+ or service.fragment
112
+ ):
113
+ raise SourceUnavailable("BIBCITE_PUBLIC_SERVICE_URL is invalid")
114
+
115
+ params = kwargs.get("params") or {}
116
+ safe_params = {
117
+ str(key): str(value)
118
+ for key, value in params.items()
119
+ if str(key).lower() not in {"api_key", "mailto"}
120
+ }
121
+ try:
122
+ response = c.post(
123
+ public_url,
124
+ json={"provider": provider, "path": target.path, "params": safe_params},
125
+ timeout=_request_timeout(timeout),
126
+ follow_redirects=False,
127
+ )
128
+ except httpx.HTTPError as e:
129
+ raise SourceUnavailable(
130
+ f"public literature service unavailable ({type(e).__name__})"
131
+ ) from e
132
+ if response.status_code == 404:
133
+ return response
134
+ if response.status_code == 429:
135
+ raise SourceUnavailable("public literature service rate-limited (429)")
136
+ if response.is_error:
137
+ raise SourceUnavailable(
138
+ f"public literature service error ({response.status_code})"
139
+ )
140
+ return response
141
+
142
+
143
+ def _sleep(delay: float):
144
+ """Sleep for pacing/backoff without crossing the publication deadline."""
145
+ deadline = getattr(_REQUEST_DEADLINE, "value", None)
146
+ if deadline is None:
147
+ time.sleep(delay)
148
+ return
149
+ remaining = deadline - time.monotonic()
150
+ if remaining <= 0:
151
+ raise PublicationTimeout("publication lookup timed out")
152
+ time.sleep(min(delay, remaining))
153
+ if delay >= remaining:
154
+ raise PublicationTimeout("publication lookup timed out")
155
+
156
+
157
+ def _with_deadline(deadline: float, fn, *args):
158
+ previous = getattr(_REQUEST_DEADLINE, "value", None)
159
+ _REQUEST_DEADLINE.value = deadline
160
+ try:
161
+ return fn(*args)
162
+ finally:
163
+ if previous is None:
164
+ del _REQUEST_DEADLINE.value
165
+ else:
166
+ _REQUEST_DEADLINE.value = previous
167
+
168
+
57
169
  def _client(browser: bool = False) -> httpx.Client:
58
170
  return httpx.Client(
59
171
  headers={"User-Agent": BROWSER_UA if browser else UA},
@@ -129,7 +241,8 @@ def arxiv_api_get(params: dict) -> httpx.Response:
129
241
  time.sleep(3 * attempt)
130
242
  try:
131
243
  with _client() as c:
132
- r = c.get(
244
+ r = _get(
245
+ c,
133
246
  "https://export.arxiv.org/api/query",
134
247
  params=params,
135
248
  timeout=30.0,
@@ -198,13 +311,13 @@ def _paced_get(
198
311
  for attempt in range(2):
199
312
  wait = min_interval - (time.monotonic() - _LAST_REQUEST.get(source, 0.0))
200
313
  if wait > 0:
201
- time.sleep(wait)
314
+ _sleep(wait)
202
315
  _LAST_REQUEST[source] = time.monotonic()
203
316
  try:
204
- r = c.get(url, params=params, headers=headers)
317
+ r = _get(c, url, params=params, headers=headers)
205
318
  except httpx.HTTPError as e: # Retry transport errors once before failing.
206
319
  if attempt < 1:
207
- time.sleep(1)
320
+ _sleep(1)
208
321
  continue
209
322
  raise TransientSourceError(
210
323
  f"{source} unreachable ({type(e).__name__})"
@@ -218,7 +331,7 @@ def _paced_get(
218
331
  # skip this source for the rest of the run.
219
332
  retry_after = int(r.headers.get("Retry-After") or 0)
220
333
  if attempt < 1 and retry_after <= 2:
221
- time.sleep(max(retry_after, 1))
334
+ _sleep(max(retry_after, 1))
222
335
  continue
223
336
  raise SourceUnavailable(f"{source} rate-limited (429)")
224
337
  return r
@@ -398,7 +511,7 @@ def arxiv_abs_metadata(arxiv_id: str) -> ArxivMeta | None:
398
511
  """Scrape the arxiv.org abs page's Highwire meta tags — the abs pages stay
399
512
  up when the export API throttles."""
400
513
  with _client(browser=True) as c:
401
- r = c.get(f"https://arxiv.org/abs/{arxiv_id}")
514
+ r = _get(c, f"https://arxiv.org/abs/{arxiv_id}")
402
515
  if r.status_code != 200:
403
516
  return None
404
517
  page = r.text
@@ -483,68 +596,14 @@ def try_semantic_scholar(
483
596
  return None
484
597
 
485
598
 
486
- # ---------------------------------------------------------------------------
487
- # Google Scholar (port of PaperMemory's background fetchGSData)
488
- # ---------------------------------------------------------------------------
489
-
490
- def try_google_scholar(title: str) -> Match | None:
491
- with _client(browser=True) as c:
492
- r = c.get(
493
- "https://scholar.google.com/scholar",
494
- params={"q": title, "hl": "en"},
495
- )
496
- if r.status_code == 429 or "captcha" in r.text.lower()[:5000]:
497
- raise SourceUnavailable("Google Scholar is blocking requests (captcha/429)")
498
- r.raise_for_status()
499
- parts = r.text.split("gs_res_ccl_mid")
500
- if len(parts) < 2:
501
- return None
502
- page = parts[1]
503
- # Each result title anchor looks like <a id="DATAID" href=...>Title</a>
504
- # (the title may contain <b> highlights and HTML entities).
505
- data_id = ""
506
- for am in re.finditer(
507
- r'<a[^>]*\bid="([\w-]{6,40})"[^>]*>(.*?)</a>', page, re.S
508
- ):
509
- text = html.unescape(re.sub(r"<[^>]+>", "", am.group(2)))
510
- if norm_title(text) == norm_title(title):
511
- data_id = am.group(1)
512
- break
513
- if not data_id:
514
- return None
515
- cite_url = (
516
- "https://scholar.google.com/scholar?q=info:"
517
- f"{data_id}:scholar.google.com/&output=cite&scirp=0&hl=en"
518
- )
519
- cite_html = c.get(cite_url).text
520
- bm = re.search(r'<a[^>]*href="([^">]+)"[^>]*>BibTex</a>', cite_html, re.I)
521
- if not bm:
522
- return None
523
- bib_url = re.sub(r"\s+", "", bm.group(1).replace("&amp;", "&"))
524
- bibtex = c.get(bib_url).text
525
- from .bibfile import parse_bibtex_entry # local import to avoid cycle
526
-
527
- entry = parse_bibtex_entry(bibtex)
528
- venue = entry.get("journal", "") or entry.get("booktitle", "")
529
- if venue and not venue.lower().endswith("xiv") and "preprint" not in venue.lower():
530
- _log(f"[googlescholar] match: {venue}")
531
- return Match(
532
- source="googlescholar",
533
- venue=venue,
534
- title=clean_title(entry.get("title", title)),
535
- year=entry.get("year", ""),
536
- bibtex=bibtex,
537
- )
538
- return None
539
-
540
-
541
599
  # ---------------------------------------------------------------------------
542
600
  # CrossRef
543
601
  # ---------------------------------------------------------------------------
544
602
 
545
603
  def try_crossref(title: str) -> Match | None:
546
604
  with _client() as c:
547
- r = c.get(
605
+ r = _get(
606
+ c,
548
607
  "https://api.crossref.org/works",
549
608
  params={
550
609
  "rows": 3,
@@ -583,8 +642,9 @@ def try_crossref(title: str) -> Match | None:
583
642
  year = str(parts[0][0])
584
643
  bibtex = ""
585
644
  if doi:
586
- br = c.get(
587
- f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex"
645
+ br = _get(
646
+ c,
647
+ f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex",
588
648
  )
589
649
  if br.status_code == 200:
590
650
  bibtex = br.text
@@ -606,7 +666,8 @@ def try_crossref(title: str) -> Match | None:
606
666
 
607
667
  def try_unpaywall(title: str) -> Match | None:
608
668
  with _client() as c:
609
- r = c.get(
669
+ r = _get(
670
+ c,
610
671
  "https://api.unpaywall.org/v2/search",
611
672
  params={"query": title, "is_oa": "true", "email": _mailto()},
612
673
  )
@@ -653,7 +714,8 @@ def try_unpaywall(title: str) -> Match | None:
653
714
  def openalex_search(title: str) -> dict | None:
654
715
  """OpenAlex work with an exactly-matching normalized title, or None."""
655
716
  with _client() as c:
656
- r = c.get(
717
+ r = _get(
718
+ c,
657
719
  "https://api.openalex.org/works",
658
720
  params=_openalex_params({"search": title, "per-page": 5}),
659
721
  )
@@ -725,7 +787,9 @@ def try_openalex(title: str) -> Match | None:
725
787
 
726
788
  def crossref_by_doi(doi: str) -> Match | None:
727
789
  with _client() as c:
728
- r = c.get(f"https://api.crossref.org/works/{doi}", params={"mailto": _mailto()})
790
+ r = _get(
791
+ c, f"https://api.crossref.org/works/{doi}", params={"mailto": _mailto()}
792
+ )
729
793
  if r.status_code != 200:
730
794
  return None
731
795
  data = r.json().get("message", {})
@@ -737,7 +801,9 @@ def crossref_by_doi(doi: str) -> Match | None:
737
801
  if parts and parts[0]:
738
802
  year = str(parts[0][0])
739
803
  bibtex = ""
740
- br = c.get(f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex")
804
+ br = _get(
805
+ c, f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex"
806
+ )
741
807
  if br.status_code == 200:
742
808
  bibtex = br.text
743
809
  authors = [
@@ -762,7 +828,6 @@ def crossref_by_doi(doi: str) -> Match | None:
762
828
  CASCADE = (
763
829
  ("dblp", lambda t, y, a, au: try_dblp(t, au)),
764
830
  ("semanticscholar", lambda t, y, a, au: try_semantic_scholar(t, y, a)),
765
- ("googlescholar", lambda t, y, a, au: try_google_scholar(t)),
766
831
  ("crossref", lambda t, y, a, au: try_crossref(t)),
767
832
  ("unpaywall", lambda t, y, a, au: try_unpaywall(t)),
768
833
  ("openalex", lambda t, y, a, au: try_openalex(t)),
@@ -775,8 +840,8 @@ CASCADE = (
775
840
  _DISABLED: dict[str, str] = {}
776
841
 
777
842
  # Only these sources are authoritative enough that losing one taints a miss
778
- # into "incomplete". Google Scholar captchas and Unpaywall flakiness are
779
- # routine and must not stop "not_found" from ever being trustworthy.
843
+ # into "incomplete". Unpaywall flakiness is routine and must not stop
844
+ # "not_found" from ever being trustworthy.
780
845
  # Override with BIBCITE_CORE_SOURCES="dblp,semanticscholar" if one of these
781
846
  # is down for days and keeps every verdict incomplete.
782
847
  CORE_SOURCES = frozenset(
@@ -815,12 +880,21 @@ def find_published(
815
880
  # version (the common case) misses everywhere, and used to pay the *sum* of
816
881
  # each source's latency; now the wall-clock is the slowest single source.
817
882
  # The first verified hit by CASCADE priority still wins.
883
+ deadline = time.monotonic() + PUBLICATION_TIMEOUT
818
884
  active = [(name, fn) for name, fn in CASCADE if name not in _DISABLED]
819
885
  outcomes: dict[str, tuple] = {}
820
886
  if active:
821
887
  with ThreadPoolExecutor(max_workers=len(active)) as pool:
822
888
  futures = {
823
- pool.submit(fn, title, year, arxiv_id, author_hint): name
889
+ pool.submit(
890
+ _with_deadline,
891
+ deadline,
892
+ fn,
893
+ title,
894
+ year,
895
+ arxiv_id,
896
+ author_hint,
897
+ ): name
824
898
  for name, fn in active
825
899
  }
826
900
  for future in as_completed(futures):
@@ -858,9 +932,18 @@ def find_published(
858
932
  # Exact-title search missed everywhere. Before concluding "no published
859
933
  # version", try the title-drift fallback — camera-ready titles frequently
860
934
  # differ from the arXiv ones, which is precisely the upgrade scenario.
861
- if author_hint and "dblp" not in _DISABLED:
935
+ # Only a clean exact DBLP miss justifies another query. A timeout or other
936
+ # failure has already spent its chance for this entry, and retrying the
937
+ # fuzzy form was doubling the worst-case interactive latency.
938
+ dblp_outcome = outcomes.get("dblp")
939
+ if (
940
+ author_hint
941
+ and dblp_outcome is not None
942
+ and dblp_outcome[0] == "miss"
943
+ and time.monotonic() < deadline
944
+ ):
862
945
  try:
863
- m = try_dblp_fuzzy(title, author_hint, year)
946
+ m = _with_deadline(deadline, try_dblp_fuzzy, title, author_hint, year)
864
947
  if m:
865
948
  cache.put(cache_key, m.__dict__)
866
949
  return m, "found"
@@ -920,7 +1003,7 @@ def fetch_web_page(url: str) -> WebPage:
920
1003
  """
921
1004
  try:
922
1005
  with _client(browser=True) as client:
923
- response = client.get(url)
1006
+ response = _get(client, url)
924
1007
  response.raise_for_status()
925
1008
  body = response.text[:400_000]
926
1009
  except httpx.HTTPStatusError as e:
@@ -2,7 +2,7 @@
2
2
 
3
3
  Parses the vendored ``data/strings.bib`` @string table (journals /
4
4
  conferences / workshops) and maps venue strings returned by DBLP, Semantic
5
- Scholar, Google Scholar, CrossRef, Unpaywall, etc. onto the canonical names.
5
+ Scholar, CrossRef, Unpaywall, etc. onto the canonical names.
6
6
  """
7
7
 
8
8
  import re
@@ -0,0 +1,69 @@
1
+ """Exercise the CLI publication cascade without contacting external services."""
2
+
3
+ import importlib
4
+ import json
5
+
6
+ import httpx
7
+ import pytest
8
+
9
+ from bibcite import cache, cli, sources
10
+
11
+ resolver = importlib.import_module("bibcite.resolve")
12
+
13
+
14
+ @pytest.mark.parametrize("operation", ["get", "add", "upgrade"])
15
+ def test_cli_never_queries_google_scholar(operation, monkeypatch, tmp_path, capsys):
16
+ monkeypatch.setattr(cache, "DISABLED", True)
17
+ monkeypatch.setattr(sources, "_DISABLED", {})
18
+ monkeypatch.setattr(
19
+ resolver,
20
+ "arxiv_metadata",
21
+ lambda _: sources.ArxivMeta(
22
+ "1706.03762",
23
+ "Attention Is All You Need",
24
+ ["Ashish Vaswani"],
25
+ "2017",
26
+ "https://arxiv.org/abs/1706.03762",
27
+ ),
28
+ )
29
+ visited = []
30
+ for name in ("dblp", "semantic_scholar", "crossref", "unpaywall", "openalex"):
31
+
32
+ def miss(*args, source=name):
33
+ visited.append(source)
34
+ return None
35
+
36
+ monkeypatch.setattr(sources, f"try_{name}", miss)
37
+ monkeypatch.setattr(sources, "try_dblp_fuzzy", lambda *args: None)
38
+ requests = []
39
+
40
+ def unexpected_request(client, url, **kwargs):
41
+ requests.append(url)
42
+ return httpx.Response(429, request=httpx.Request("GET", url))
43
+
44
+ monkeypatch.setattr(sources, "_get", unexpected_request)
45
+ path = tmp_path / "references.bib"
46
+ if operation == "get":
47
+ args = ["get", "--json", "1706.03762"]
48
+ elif operation == "add":
49
+ args = ["add", "--no-tidy", str(path), "1706.03762"]
50
+ else:
51
+ path.write_text("""@article{vaswani2017attention,
52
+ title = {Attention Is All You Need},
53
+ author = {Ashish Vaswani},
54
+ year = {2017},
55
+ journal = {arXiv preprint arXiv:1706.03762},
56
+ eprint = {1706.03762}
57
+ }
58
+ """)
59
+ args = ["upgrade", str(path), "--dry-run"]
60
+ cli.main(args)
61
+ json.loads(capsys.readouterr().out)
62
+ assert set(visited) == {
63
+ "dblp",
64
+ "semantic_scholar",
65
+ "crossref",
66
+ "unpaywall",
67
+ "openalex",
68
+ }
69
+ assert requests == []
@@ -0,0 +1,188 @@
1
+ import json
2
+
3
+ import httpx
4
+ import pytest
5
+
6
+ from bibcite import sources
7
+
8
+
9
+ @pytest.fixture(autouse=True)
10
+ def clean_environment(monkeypatch):
11
+ for name in (
12
+ "BIBCITE_PUBLIC_SERVICE_URL",
13
+ "BIBCITE_MAILTO",
14
+ "OPENALEX_API_KEY",
15
+ "S2_API_KEY",
16
+ "SEMANTIC_SCHOLAR_API_KEY",
17
+ ):
18
+ monkeypatch.delenv(name, raising=False)
19
+ monkeypatch.setattr(sources, "_LAST_REQUEST", {})
20
+ monkeypatch.setattr(sources.time, "sleep", lambda _: None)
21
+
22
+
23
+ def _client(handler):
24
+ return httpx.Client(transport=httpx.MockTransport(handler))
25
+
26
+
27
+ @pytest.mark.parametrize(
28
+ ("url", "provider", "path"),
29
+ (
30
+ ("https://api.openalex.org/works", "openalex", "/works"),
31
+ (
32
+ "https://api.semanticscholar.org/graph/v1/paper/search",
33
+ "semanticscholar",
34
+ "/graph/v1/paper/search",
35
+ ),
36
+ (
37
+ "https://api.crossref.org/works/10.1234/example/transform/application/x-bibtex",
38
+ "crossref",
39
+ "/works/10.1234/example/transform/application/x-bibtex",
40
+ ),
41
+ ),
42
+ )
43
+ def test_keyless_supported_requests_route_to_public_service(
44
+ url, provider, path, monkeypatch
45
+ ):
46
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
47
+ requests = []
48
+
49
+ def handler(request):
50
+ requests.append(request)
51
+ return httpx.Response(200, json={"unchanged": True})
52
+
53
+ with _client(handler) as client:
54
+ response = sources._get(
55
+ client,
56
+ url,
57
+ params={
58
+ "query": "paper",
59
+ "limit": 5,
60
+ "api_key": "secret",
61
+ "mailto": "secret@example.com",
62
+ },
63
+ headers={"x-api-key": "secret"},
64
+ )
65
+
66
+ assert response.json() == {"unchanged": True}
67
+ assert len(requests) == 1
68
+ request = requests[0]
69
+ assert request.method == "POST"
70
+ assert request.url == "https://literature.test/v1/query"
71
+ assert "x-api-key" not in request.headers
72
+ payload = json.loads(request.content)
73
+ assert payload == {
74
+ "provider": provider,
75
+ "path": path,
76
+ "params": {"query": "paper", "limit": "5"},
77
+ }
78
+ assert "secret" not in request.content.decode()
79
+
80
+
81
+ @pytest.mark.parametrize(
82
+ ("variable", "url", "params"),
83
+ (
84
+ ("OPENALEX_API_KEY", "https://api.openalex.org/works", {"api_key": "mine"}),
85
+ ("S2_API_KEY", "https://api.semanticscholar.org/graph/v1/paper/search", {}),
86
+ (
87
+ "BIBCITE_MAILTO",
88
+ "https://api.crossref.org/works",
89
+ {"mailto": "me@example.com"},
90
+ ),
91
+ ),
92
+ )
93
+ def test_personal_access_overrides_public_service(variable, url, params, monkeypatch):
94
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
95
+ monkeypatch.setenv(variable, "mine")
96
+ requests = []
97
+
98
+ def handler(request):
99
+ requests.append(request)
100
+ return httpx.Response(200, json={})
101
+
102
+ with _client(handler) as client:
103
+ sources._get(client, url, params=params)
104
+
105
+ assert len(requests) == 1
106
+ assert requests[0].method == "GET"
107
+ assert requests[0].url.host != "literature.test"
108
+
109
+
110
+ def test_without_public_service_requests_remain_direct():
111
+ requests = []
112
+
113
+ def handler(request):
114
+ requests.append(request)
115
+ return httpx.Response(200, json={})
116
+
117
+ with _client(handler) as client:
118
+ sources._get(client, "https://api.openalex.org/works", params={"search": "x"})
119
+
120
+ assert len(requests) == 1
121
+ assert requests[0].method == "GET"
122
+ assert requests[0].url.host == "api.openalex.org"
123
+
124
+
125
+ @pytest.mark.parametrize("status", [429, 500])
126
+ def test_public_service_http_failure_is_not_retried(status, monkeypatch):
127
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
128
+ calls = 0
129
+
130
+ def handler(request):
131
+ nonlocal calls
132
+ calls += 1
133
+ return httpx.Response(status)
134
+
135
+ with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
136
+ sources._paced_get(
137
+ client,
138
+ "https://api.semanticscholar.org/graph/v1/paper/search",
139
+ "semanticscholar",
140
+ 0,
141
+ )
142
+
143
+ assert calls == 1
144
+
145
+
146
+ def test_public_service_timeout_is_not_retried(monkeypatch):
147
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
148
+ calls = 0
149
+
150
+ def handler(request):
151
+ nonlocal calls
152
+ calls += 1
153
+ raise httpx.ReadTimeout("timed out", request=request)
154
+
155
+ with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
156
+ sources._paced_get(
157
+ client,
158
+ "https://api.semanticscholar.org/graph/v1/paper/search",
159
+ "semanticscholar",
160
+ 0,
161
+ )
162
+
163
+ assert calls == 1
164
+
165
+
166
+ def test_public_service_404_is_preserved(monkeypatch):
167
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
168
+
169
+ with _client(lambda request: httpx.Response(404)) as client:
170
+ response = sources._get(client, "https://api.openalex.org/works/W1")
171
+
172
+ assert response.status_code == 404
173
+
174
+
175
+ @pytest.mark.parametrize(
176
+ "value",
177
+ (
178
+ "http://literature.test/v1/query",
179
+ "https://user:pass@literature.test/v1/query",
180
+ "https://literature.test/v1/query?secret=x",
181
+ "https://literature.test/v1/query#fragment",
182
+ ),
183
+ )
184
+ def test_invalid_public_service_url_is_rejected(value, monkeypatch):
185
+ monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", value)
186
+ with _client(lambda request: httpx.Response(200)) as client:
187
+ with pytest.raises(sources.SourceUnavailable):
188
+ sources._get(client, "https://api.openalex.org/works")
@@ -5,7 +5,12 @@ import bibcite.sources as sources
5
5
  from bibcite import cache
6
6
  from bibcite.bibfile import load_bib_file
7
7
  from bibcite.cli import _upgrade_entries
8
- from bibcite.sources import Match, SourceUnavailable, find_published
8
+ from bibcite.sources import (
9
+ Match,
10
+ SourceUnavailable,
11
+ TransientSourceError,
12
+ find_published,
13
+ )
9
14
 
10
15
 
11
16
  @pytest.fixture(autouse=True)
@@ -22,7 +27,7 @@ class _ReadErrorClient:
22
27
  self.failures = failures
23
28
  self.calls = 0
24
29
 
25
- def get(self, url, params=None, headers=None):
30
+ def get(self, url, params=None, headers=None, timeout=None):
26
31
  self.calls += 1
27
32
  request = httpx.Request("GET", url, params=params, headers=headers)
28
33
  if self.calls <= self.failures:
@@ -70,6 +75,41 @@ def test_dblp_read_failures_do_not_disable_later_batch_entries(monkeypatch):
70
75
  assert second_match.venue == "TMLR"
71
76
 
72
77
 
78
+ def test_dblp_transport_failure_skips_the_fuzzy_retry(monkeypatch):
79
+ fuzzy_calls = 0
80
+
81
+ def dblp(*args):
82
+ raise TransientSourceError("simulated timeout")
83
+
84
+ def fuzzy(*args):
85
+ nonlocal fuzzy_calls
86
+ fuzzy_calls += 1
87
+
88
+ monkeypatch.setattr(
89
+ sources,
90
+ "CASCADE",
91
+ (
92
+ ("dblp", dblp),
93
+ ("crossref", lambda *args: None),
94
+ ),
95
+ )
96
+ monkeypatch.setattr(sources, "try_dblp_fuzzy", fuzzy)
97
+
98
+ match, status = find_published("First paper", author_hint="doe")
99
+
100
+ assert (match, status) == (None, "incomplete")
101
+ assert fuzzy_calls == 0
102
+
103
+
104
+ def test_requests_use_only_the_remaining_publication_budget(monkeypatch):
105
+ monkeypatch.setattr(sources.time, "monotonic", lambda: 7.0)
106
+ sources._REQUEST_DEADLINE.value = 10.0
107
+ try:
108
+ assert sources._request_timeout(8.0) == 3.0
109
+ finally:
110
+ del sources._REQUEST_DEADLINE.value
111
+
112
+
73
113
  def test_upgrade_retries_dblp_after_previous_entry_read_failures(
74
114
  tmp_path, monkeypatch
75
115
  ):
@@ -43,7 +43,7 @@ def test_core_source_429_taints_verdict(monkeypatch):
43
43
  monkeypatch.setattr(
44
44
  sources,
45
45
  "CASCADE",
46
- _cascade(dblp="raise", googlescholar=None, crossref=None),
46
+ _cascade(dblp="raise", unpaywall=None, crossref=None),
47
47
  )
48
48
  match, status = find_published("Some Title", author_hint="smith")
49
49
  assert (match, status) == (None, "incomplete")
@@ -60,11 +60,11 @@ def test_previously_disabled_core_source_taints_next_queries(monkeypatch):
60
60
 
61
61
 
62
62
  def test_noncore_outage_does_not_taint(monkeypatch):
63
- # Google Scholar captcha is routine; a miss stays trustworthy.
63
+ # Unpaywall outages are routine; a miss stays trustworthy.
64
64
  monkeypatch.setattr(
65
65
  sources,
66
66
  "CASCADE",
67
- _cascade(dblp=None, googlescholar="raise", crossref=None),
67
+ _cascade(dblp=None, unpaywall="raise", crossref=None),
68
68
  )
69
69
  match, status = find_published("Some Title", author_hint="smith")
70
70
  assert (match, status) == (None, "not_found")
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.6.2"
21
+ version = "0.6.4"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes