bibcite-cli 0.6.2__tar.gz → 0.6.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/PKG-INFO +2 -2
  2. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/pyproject.toml +1 -1
  3. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/sources.py +103 -20
  4. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_source_retries.py +42 -2
  5. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/uv.lock +1 -1
  6. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/.github/workflows/ci.yml +0 -0
  7. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/.github/workflows/publish.yml +0 -0
  8. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/.gitignore +0 -0
  9. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/LICENSE +0 -0
  10. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/Readme.md +0 -0
  11. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/assets/bibcite.svg +0 -0
  12. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/skills/bibcite/SKILL.md +0 -0
  13. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/__init__.py +0 -0
  14. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/bibfile.py +0 -0
  15. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/cache.py +0 -0
  16. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/cli.py +0 -0
  17. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/data/strings.bib +0 -0
  18. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/normalize.py +0 -0
  19. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/resolve.py +0 -0
  20. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/venues.py +0 -0
  21. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_bibfile.py +0 -0
  22. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_bugfixes.py +0 -0
  23. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_cli_status.py +0 -0
  24. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_entry_types.py +0 -0
  25. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_normalize.py +0 -0
  26. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_round2.py +0 -0
  27. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_round3.py +0 -0
  28. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_status_semantics.py +0 -0
  29. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_strings_override.py +0 -0
  30. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_venues.py +0 -0
  31. {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_webpages.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: bibcite-cli
3
- Version: 0.6.2
3
+ Version: 0.6.3
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.6.2"
3
+ version = "0.6.3"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -10,6 +10,7 @@ import html
10
10
  import os
11
11
  import re
12
12
  import sys
13
+ import threading
13
14
  import time
14
15
  import xml.etree.ElementTree as ET
15
16
  from concurrent.futures import ThreadPoolExecutor, as_completed
@@ -30,6 +31,10 @@ BROWSER_UA = (
30
31
  # never a false "not published". The arXiv metadata fetch sets its own longer
31
32
  # timeout on the request itself, so this does not affect it.
32
33
  TIMEOUT = 8.0
34
+ # Publication matching is enrichment on top of a valid arXiv citation. Keep
35
+ # the entire concurrent cascade, including its DBLP title-drift fallback,
36
+ # within an interactive budget instead of letting sequential retries add up.
37
+ PUBLICATION_TIMEOUT = 10.0
33
38
 
34
39
  PREPRINT_VENUES = re.compile(r"arxiv|corr|biorxiv|medrxiv|chemrxiv|ssrn|preprint", re.I)
35
40
  ARXIV_DOI = re.compile(r"^10\.48550/", re.I)
@@ -54,6 +59,56 @@ class TransientSourceError(SourceUnavailable):
54
59
  process-wide circuit breaker for later batch entries."""
55
60
 
56
61
 
62
+ class PublicationTimeout(TransientSourceError):
63
+ """The total publication-matching budget was exhausted."""
64
+
65
+
66
+ _REQUEST_DEADLINE = threading.local()
67
+
68
+
69
+ def _request_timeout(cap: float = TIMEOUT) -> float:
70
+ deadline = getattr(_REQUEST_DEADLINE, "value", None)
71
+ if deadline is None:
72
+ return cap
73
+ remaining = deadline - time.monotonic()
74
+ if remaining <= 0:
75
+ raise PublicationTimeout("publication lookup timed out")
76
+ return max(0.001, min(cap, remaining))
77
+
78
+
79
+ def _get(
80
+ c: httpx.Client, url: str, *, timeout: float = TIMEOUT, **kwargs
81
+ ) -> httpx.Response:
82
+ """GET with the per-request cap narrowed by the active total deadline."""
83
+ return c.get(url, timeout=_request_timeout(timeout), **kwargs)
84
+
85
+
86
+ def _sleep(delay: float):
87
+ """Sleep for pacing/backoff without crossing the publication deadline."""
88
+ deadline = getattr(_REQUEST_DEADLINE, "value", None)
89
+ if deadline is None:
90
+ time.sleep(delay)
91
+ return
92
+ remaining = deadline - time.monotonic()
93
+ if remaining <= 0:
94
+ raise PublicationTimeout("publication lookup timed out")
95
+ time.sleep(min(delay, remaining))
96
+ if delay >= remaining:
97
+ raise PublicationTimeout("publication lookup timed out")
98
+
99
+
100
+ def _with_deadline(deadline: float, fn, *args):
101
+ previous = getattr(_REQUEST_DEADLINE, "value", None)
102
+ _REQUEST_DEADLINE.value = deadline
103
+ try:
104
+ return fn(*args)
105
+ finally:
106
+ if previous is None:
107
+ del _REQUEST_DEADLINE.value
108
+ else:
109
+ _REQUEST_DEADLINE.value = previous
110
+
111
+
57
112
  def _client(browser: bool = False) -> httpx.Client:
58
113
  return httpx.Client(
59
114
  headers={"User-Agent": BROWSER_UA if browser else UA},
@@ -129,7 +184,8 @@ def arxiv_api_get(params: dict) -> httpx.Response:
129
184
  time.sleep(3 * attempt)
130
185
  try:
131
186
  with _client() as c:
132
- r = c.get(
187
+ r = _get(
188
+ c,
133
189
  "https://export.arxiv.org/api/query",
134
190
  params=params,
135
191
  timeout=30.0,
@@ -198,13 +254,13 @@ def _paced_get(
198
254
  for attempt in range(2):
199
255
  wait = min_interval - (time.monotonic() - _LAST_REQUEST.get(source, 0.0))
200
256
  if wait > 0:
201
- time.sleep(wait)
257
+ _sleep(wait)
202
258
  _LAST_REQUEST[source] = time.monotonic()
203
259
  try:
204
- r = c.get(url, params=params, headers=headers)
260
+ r = _get(c, url, params=params, headers=headers)
205
261
  except httpx.HTTPError as e: # Retry transport errors once before failing.
206
262
  if attempt < 1:
207
- time.sleep(1)
263
+ _sleep(1)
208
264
  continue
209
265
  raise TransientSourceError(
210
266
  f"{source} unreachable ({type(e).__name__})"
@@ -218,7 +274,7 @@ def _paced_get(
218
274
  # skip this source for the rest of the run.
219
275
  retry_after = int(r.headers.get("Retry-After") or 0)
220
276
  if attempt < 1 and retry_after <= 2:
221
- time.sleep(max(retry_after, 1))
277
+ _sleep(max(retry_after, 1))
222
278
  continue
223
279
  raise SourceUnavailable(f"{source} rate-limited (429)")
224
280
  return r
@@ -398,7 +454,7 @@ def arxiv_abs_metadata(arxiv_id: str) -> ArxivMeta | None:
398
454
  """Scrape the arxiv.org abs page's Highwire meta tags — the abs pages stay
399
455
  up when the export API throttles."""
400
456
  with _client(browser=True) as c:
401
- r = c.get(f"https://arxiv.org/abs/{arxiv_id}")
457
+ r = _get(c, f"https://arxiv.org/abs/{arxiv_id}")
402
458
  if r.status_code != 200:
403
459
  return None
404
460
  page = r.text
@@ -489,7 +545,8 @@ def try_semantic_scholar(
489
545
 
490
546
  def try_google_scholar(title: str) -> Match | None:
491
547
  with _client(browser=True) as c:
492
- r = c.get(
548
+ r = _get(
549
+ c,
493
550
  "https://scholar.google.com/scholar",
494
551
  params={"q": title, "hl": "en"},
495
552
  )
@@ -516,12 +573,12 @@ def try_google_scholar(title: str) -> Match | None:
516
573
  "https://scholar.google.com/scholar?q=info:"
517
574
  f"{data_id}:scholar.google.com/&output=cite&scirp=0&hl=en"
518
575
  )
519
- cite_html = c.get(cite_url).text
576
+ cite_html = _get(c, cite_url).text
520
577
  bm = re.search(r'<a[^>]*href="([^">]+)"[^>]*>BibTex</a>', cite_html, re.I)
521
578
  if not bm:
522
579
  return None
523
580
  bib_url = re.sub(r"\s+", "", bm.group(1).replace("&amp;", "&"))
524
- bibtex = c.get(bib_url).text
581
+ bibtex = _get(c, bib_url).text
525
582
  from .bibfile import parse_bibtex_entry # local import to avoid cycle
526
583
 
527
584
  entry = parse_bibtex_entry(bibtex)
@@ -544,7 +601,8 @@ def try_google_scholar(title: str) -> Match | None:
544
601
 
545
602
  def try_crossref(title: str) -> Match | None:
546
603
  with _client() as c:
547
- r = c.get(
604
+ r = _get(
605
+ c,
548
606
  "https://api.crossref.org/works",
549
607
  params={
550
608
  "rows": 3,
@@ -583,8 +641,9 @@ def try_crossref(title: str) -> Match | None:
583
641
  year = str(parts[0][0])
584
642
  bibtex = ""
585
643
  if doi:
586
- br = c.get(
587
- f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex"
644
+ br = _get(
645
+ c,
646
+ f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex",
588
647
  )
589
648
  if br.status_code == 200:
590
649
  bibtex = br.text
@@ -606,7 +665,8 @@ def try_crossref(title: str) -> Match | None:
606
665
 
607
666
  def try_unpaywall(title: str) -> Match | None:
608
667
  with _client() as c:
609
- r = c.get(
668
+ r = _get(
669
+ c,
610
670
  "https://api.unpaywall.org/v2/search",
611
671
  params={"query": title, "is_oa": "true", "email": _mailto()},
612
672
  )
@@ -653,7 +713,8 @@ def try_unpaywall(title: str) -> Match | None:
653
713
  def openalex_search(title: str) -> dict | None:
654
714
  """OpenAlex work with an exactly-matching normalized title, or None."""
655
715
  with _client() as c:
656
- r = c.get(
716
+ r = _get(
717
+ c,
657
718
  "https://api.openalex.org/works",
658
719
  params=_openalex_params({"search": title, "per-page": 5}),
659
720
  )
@@ -725,7 +786,9 @@ def try_openalex(title: str) -> Match | None:
725
786
 
726
787
  def crossref_by_doi(doi: str) -> Match | None:
727
788
  with _client() as c:
728
- r = c.get(f"https://api.crossref.org/works/{doi}", params={"mailto": _mailto()})
789
+ r = _get(
790
+ c, f"https://api.crossref.org/works/{doi}", params={"mailto": _mailto()}
791
+ )
729
792
  if r.status_code != 200:
730
793
  return None
731
794
  data = r.json().get("message", {})
@@ -737,7 +800,9 @@ def crossref_by_doi(doi: str) -> Match | None:
737
800
  if parts and parts[0]:
738
801
  year = str(parts[0][0])
739
802
  bibtex = ""
740
- br = c.get(f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex")
803
+ br = _get(
804
+ c, f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex"
805
+ )
741
806
  if br.status_code == 200:
742
807
  bibtex = br.text
743
808
  authors = [
@@ -815,12 +880,21 @@ def find_published(
815
880
  # version (the common case) misses everywhere, and used to pay the *sum* of
816
881
  # each source's latency; now the wall-clock is the slowest single source.
817
882
  # The first verified hit by CASCADE priority still wins.
883
+ deadline = time.monotonic() + PUBLICATION_TIMEOUT
818
884
  active = [(name, fn) for name, fn in CASCADE if name not in _DISABLED]
819
885
  outcomes: dict[str, tuple] = {}
820
886
  if active:
821
887
  with ThreadPoolExecutor(max_workers=len(active)) as pool:
822
888
  futures = {
823
- pool.submit(fn, title, year, arxiv_id, author_hint): name
889
+ pool.submit(
890
+ _with_deadline,
891
+ deadline,
892
+ fn,
893
+ title,
894
+ year,
895
+ arxiv_id,
896
+ author_hint,
897
+ ): name
824
898
  for name, fn in active
825
899
  }
826
900
  for future in as_completed(futures):
@@ -858,9 +932,18 @@ def find_published(
858
932
  # Exact-title search missed everywhere. Before concluding "no published
859
933
  # version", try the title-drift fallback — camera-ready titles frequently
860
934
  # differ from the arXiv ones, which is precisely the upgrade scenario.
861
- if author_hint and "dblp" not in _DISABLED:
935
+ # Only a clean exact DBLP miss justifies another query. A timeout or other
936
+ # failure has already spent its chance for this entry, and retrying the
937
+ # fuzzy form was doubling the worst-case interactive latency.
938
+ dblp_outcome = outcomes.get("dblp")
939
+ if (
940
+ author_hint
941
+ and dblp_outcome is not None
942
+ and dblp_outcome[0] == "miss"
943
+ and time.monotonic() < deadline
944
+ ):
862
945
  try:
863
- m = try_dblp_fuzzy(title, author_hint, year)
946
+ m = _with_deadline(deadline, try_dblp_fuzzy, title, author_hint, year)
864
947
  if m:
865
948
  cache.put(cache_key, m.__dict__)
866
949
  return m, "found"
@@ -920,7 +1003,7 @@ def fetch_web_page(url: str) -> WebPage:
920
1003
  """
921
1004
  try:
922
1005
  with _client(browser=True) as client:
923
- response = client.get(url)
1006
+ response = _get(client, url)
924
1007
  response.raise_for_status()
925
1008
  body = response.text[:400_000]
926
1009
  except httpx.HTTPStatusError as e:
@@ -5,7 +5,12 @@ import bibcite.sources as sources
5
5
  from bibcite import cache
6
6
  from bibcite.bibfile import load_bib_file
7
7
  from bibcite.cli import _upgrade_entries
8
- from bibcite.sources import Match, SourceUnavailable, find_published
8
+ from bibcite.sources import (
9
+ Match,
10
+ SourceUnavailable,
11
+ TransientSourceError,
12
+ find_published,
13
+ )
9
14
 
10
15
 
11
16
  @pytest.fixture(autouse=True)
@@ -22,7 +27,7 @@ class _ReadErrorClient:
22
27
  self.failures = failures
23
28
  self.calls = 0
24
29
 
25
- def get(self, url, params=None, headers=None):
30
+ def get(self, url, params=None, headers=None, timeout=None):
26
31
  self.calls += 1
27
32
  request = httpx.Request("GET", url, params=params, headers=headers)
28
33
  if self.calls <= self.failures:
@@ -70,6 +75,41 @@ def test_dblp_read_failures_do_not_disable_later_batch_entries(monkeypatch):
70
75
  assert second_match.venue == "TMLR"
71
76
 
72
77
 
78
+ def test_dblp_transport_failure_skips_the_fuzzy_retry(monkeypatch):
79
+ fuzzy_calls = 0
80
+
81
+ def dblp(*args):
82
+ raise TransientSourceError("simulated timeout")
83
+
84
+ def fuzzy(*args):
85
+ nonlocal fuzzy_calls
86
+ fuzzy_calls += 1
87
+
88
+ monkeypatch.setattr(
89
+ sources,
90
+ "CASCADE",
91
+ (
92
+ ("dblp", dblp),
93
+ ("crossref", lambda *args: None),
94
+ ),
95
+ )
96
+ monkeypatch.setattr(sources, "try_dblp_fuzzy", fuzzy)
97
+
98
+ match, status = find_published("First paper", author_hint="doe")
99
+
100
+ assert (match, status) == (None, "incomplete")
101
+ assert fuzzy_calls == 0
102
+
103
+
104
+ def test_requests_use_only_the_remaining_publication_budget(monkeypatch):
105
+ monkeypatch.setattr(sources.time, "monotonic", lambda: 7.0)
106
+ sources._REQUEST_DEADLINE.value = 10.0
107
+ try:
108
+ assert sources._request_timeout(8.0) == 3.0
109
+ finally:
110
+ del sources._REQUEST_DEADLINE.value
111
+
112
+
73
113
  def test_upgrade_retries_dblp_after_previous_entry_read_failures(
74
114
  tmp_path, monkeypatch
75
115
  ):
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.6.2"
21
+ version = "0.6.3"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes
File without changes