bibcite-cli 0.6.2__tar.gz → 0.6.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/PKG-INFO +2 -2
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/pyproject.toml +1 -1
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/sources.py +103 -20
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_source_retries.py +42 -2
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/uv.lock +1 -1
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/.github/workflows/ci.yml +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/.github/workflows/publish.yml +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/.gitignore +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/LICENSE +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/Readme.md +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/assets/bibcite.svg +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/skills/bibcite/SKILL.md +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/__init__.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/bibfile.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/cli.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/normalize.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/resolve.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/src/bibcite/venues.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_cli_status.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_round2.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_round3.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_status_semantics.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_venues.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.3}/tests/test_webpages.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.3
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -10,6 +10,7 @@ import html
|
|
|
10
10
|
import os
|
|
11
11
|
import re
|
|
12
12
|
import sys
|
|
13
|
+
import threading
|
|
13
14
|
import time
|
|
14
15
|
import xml.etree.ElementTree as ET
|
|
15
16
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
@@ -30,6 +31,10 @@ BROWSER_UA = (
|
|
|
30
31
|
# never a false "not published". The arXiv metadata fetch sets its own longer
|
|
31
32
|
# timeout on the request itself, so this does not affect it.
|
|
32
33
|
TIMEOUT = 8.0
|
|
34
|
+
# Publication matching is enrichment on top of a valid arXiv citation. Keep
|
|
35
|
+
# the entire concurrent cascade, including its DBLP title-drift fallback,
|
|
36
|
+
# within an interactive budget instead of letting sequential retries add up.
|
|
37
|
+
PUBLICATION_TIMEOUT = 10.0
|
|
33
38
|
|
|
34
39
|
PREPRINT_VENUES = re.compile(r"arxiv|corr|biorxiv|medrxiv|chemrxiv|ssrn|preprint", re.I)
|
|
35
40
|
ARXIV_DOI = re.compile(r"^10\.48550/", re.I)
|
|
@@ -54,6 +59,56 @@ class TransientSourceError(SourceUnavailable):
|
|
|
54
59
|
process-wide circuit breaker for later batch entries."""
|
|
55
60
|
|
|
56
61
|
|
|
62
|
+
class PublicationTimeout(TransientSourceError):
|
|
63
|
+
"""The total publication-matching budget was exhausted."""
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
_REQUEST_DEADLINE = threading.local()
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _request_timeout(cap: float = TIMEOUT) -> float:
|
|
70
|
+
deadline = getattr(_REQUEST_DEADLINE, "value", None)
|
|
71
|
+
if deadline is None:
|
|
72
|
+
return cap
|
|
73
|
+
remaining = deadline - time.monotonic()
|
|
74
|
+
if remaining <= 0:
|
|
75
|
+
raise PublicationTimeout("publication lookup timed out")
|
|
76
|
+
return max(0.001, min(cap, remaining))
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _get(
|
|
80
|
+
c: httpx.Client, url: str, *, timeout: float = TIMEOUT, **kwargs
|
|
81
|
+
) -> httpx.Response:
|
|
82
|
+
"""GET with the per-request cap narrowed by the active total deadline."""
|
|
83
|
+
return c.get(url, timeout=_request_timeout(timeout), **kwargs)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _sleep(delay: float):
|
|
87
|
+
"""Sleep for pacing/backoff without crossing the publication deadline."""
|
|
88
|
+
deadline = getattr(_REQUEST_DEADLINE, "value", None)
|
|
89
|
+
if deadline is None:
|
|
90
|
+
time.sleep(delay)
|
|
91
|
+
return
|
|
92
|
+
remaining = deadline - time.monotonic()
|
|
93
|
+
if remaining <= 0:
|
|
94
|
+
raise PublicationTimeout("publication lookup timed out")
|
|
95
|
+
time.sleep(min(delay, remaining))
|
|
96
|
+
if delay >= remaining:
|
|
97
|
+
raise PublicationTimeout("publication lookup timed out")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _with_deadline(deadline: float, fn, *args):
|
|
101
|
+
previous = getattr(_REQUEST_DEADLINE, "value", None)
|
|
102
|
+
_REQUEST_DEADLINE.value = deadline
|
|
103
|
+
try:
|
|
104
|
+
return fn(*args)
|
|
105
|
+
finally:
|
|
106
|
+
if previous is None:
|
|
107
|
+
del _REQUEST_DEADLINE.value
|
|
108
|
+
else:
|
|
109
|
+
_REQUEST_DEADLINE.value = previous
|
|
110
|
+
|
|
111
|
+
|
|
57
112
|
def _client(browser: bool = False) -> httpx.Client:
|
|
58
113
|
return httpx.Client(
|
|
59
114
|
headers={"User-Agent": BROWSER_UA if browser else UA},
|
|
@@ -129,7 +184,8 @@ def arxiv_api_get(params: dict) -> httpx.Response:
|
|
|
129
184
|
time.sleep(3 * attempt)
|
|
130
185
|
try:
|
|
131
186
|
with _client() as c:
|
|
132
|
-
r =
|
|
187
|
+
r = _get(
|
|
188
|
+
c,
|
|
133
189
|
"https://export.arxiv.org/api/query",
|
|
134
190
|
params=params,
|
|
135
191
|
timeout=30.0,
|
|
@@ -198,13 +254,13 @@ def _paced_get(
|
|
|
198
254
|
for attempt in range(2):
|
|
199
255
|
wait = min_interval - (time.monotonic() - _LAST_REQUEST.get(source, 0.0))
|
|
200
256
|
if wait > 0:
|
|
201
|
-
|
|
257
|
+
_sleep(wait)
|
|
202
258
|
_LAST_REQUEST[source] = time.monotonic()
|
|
203
259
|
try:
|
|
204
|
-
r = c
|
|
260
|
+
r = _get(c, url, params=params, headers=headers)
|
|
205
261
|
except httpx.HTTPError as e: # Retry transport errors once before failing.
|
|
206
262
|
if attempt < 1:
|
|
207
|
-
|
|
263
|
+
_sleep(1)
|
|
208
264
|
continue
|
|
209
265
|
raise TransientSourceError(
|
|
210
266
|
f"{source} unreachable ({type(e).__name__})"
|
|
@@ -218,7 +274,7 @@ def _paced_get(
|
|
|
218
274
|
# skip this source for the rest of the run.
|
|
219
275
|
retry_after = int(r.headers.get("Retry-After") or 0)
|
|
220
276
|
if attempt < 1 and retry_after <= 2:
|
|
221
|
-
|
|
277
|
+
_sleep(max(retry_after, 1))
|
|
222
278
|
continue
|
|
223
279
|
raise SourceUnavailable(f"{source} rate-limited (429)")
|
|
224
280
|
return r
|
|
@@ -398,7 +454,7 @@ def arxiv_abs_metadata(arxiv_id: str) -> ArxivMeta | None:
|
|
|
398
454
|
"""Scrape the arxiv.org abs page's Highwire meta tags — the abs pages stay
|
|
399
455
|
up when the export API throttles."""
|
|
400
456
|
with _client(browser=True) as c:
|
|
401
|
-
r = c
|
|
457
|
+
r = _get(c, f"https://arxiv.org/abs/{arxiv_id}")
|
|
402
458
|
if r.status_code != 200:
|
|
403
459
|
return None
|
|
404
460
|
page = r.text
|
|
@@ -489,7 +545,8 @@ def try_semantic_scholar(
|
|
|
489
545
|
|
|
490
546
|
def try_google_scholar(title: str) -> Match | None:
|
|
491
547
|
with _client(browser=True) as c:
|
|
492
|
-
r =
|
|
548
|
+
r = _get(
|
|
549
|
+
c,
|
|
493
550
|
"https://scholar.google.com/scholar",
|
|
494
551
|
params={"q": title, "hl": "en"},
|
|
495
552
|
)
|
|
@@ -516,12 +573,12 @@ def try_google_scholar(title: str) -> Match | None:
|
|
|
516
573
|
"https://scholar.google.com/scholar?q=info:"
|
|
517
574
|
f"{data_id}:scholar.google.com/&output=cite&scirp=0&hl=en"
|
|
518
575
|
)
|
|
519
|
-
cite_html = c
|
|
576
|
+
cite_html = _get(c, cite_url).text
|
|
520
577
|
bm = re.search(r'<a[^>]*href="([^">]+)"[^>]*>BibTex</a>', cite_html, re.I)
|
|
521
578
|
if not bm:
|
|
522
579
|
return None
|
|
523
580
|
bib_url = re.sub(r"\s+", "", bm.group(1).replace("&", "&"))
|
|
524
|
-
bibtex = c
|
|
581
|
+
bibtex = _get(c, bib_url).text
|
|
525
582
|
from .bibfile import parse_bibtex_entry # local import to avoid cycle
|
|
526
583
|
|
|
527
584
|
entry = parse_bibtex_entry(bibtex)
|
|
@@ -544,7 +601,8 @@ def try_google_scholar(title: str) -> Match | None:
|
|
|
544
601
|
|
|
545
602
|
def try_crossref(title: str) -> Match | None:
|
|
546
603
|
with _client() as c:
|
|
547
|
-
r =
|
|
604
|
+
r = _get(
|
|
605
|
+
c,
|
|
548
606
|
"https://api.crossref.org/works",
|
|
549
607
|
params={
|
|
550
608
|
"rows": 3,
|
|
@@ -583,8 +641,9 @@ def try_crossref(title: str) -> Match | None:
|
|
|
583
641
|
year = str(parts[0][0])
|
|
584
642
|
bibtex = ""
|
|
585
643
|
if doi:
|
|
586
|
-
br =
|
|
587
|
-
|
|
644
|
+
br = _get(
|
|
645
|
+
c,
|
|
646
|
+
f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex",
|
|
588
647
|
)
|
|
589
648
|
if br.status_code == 200:
|
|
590
649
|
bibtex = br.text
|
|
@@ -606,7 +665,8 @@ def try_crossref(title: str) -> Match | None:
|
|
|
606
665
|
|
|
607
666
|
def try_unpaywall(title: str) -> Match | None:
|
|
608
667
|
with _client() as c:
|
|
609
|
-
r =
|
|
668
|
+
r = _get(
|
|
669
|
+
c,
|
|
610
670
|
"https://api.unpaywall.org/v2/search",
|
|
611
671
|
params={"query": title, "is_oa": "true", "email": _mailto()},
|
|
612
672
|
)
|
|
@@ -653,7 +713,8 @@ def try_unpaywall(title: str) -> Match | None:
|
|
|
653
713
|
def openalex_search(title: str) -> dict | None:
|
|
654
714
|
"""OpenAlex work with an exactly-matching normalized title, or None."""
|
|
655
715
|
with _client() as c:
|
|
656
|
-
r =
|
|
716
|
+
r = _get(
|
|
717
|
+
c,
|
|
657
718
|
"https://api.openalex.org/works",
|
|
658
719
|
params=_openalex_params({"search": title, "per-page": 5}),
|
|
659
720
|
)
|
|
@@ -725,7 +786,9 @@ def try_openalex(title: str) -> Match | None:
|
|
|
725
786
|
|
|
726
787
|
def crossref_by_doi(doi: str) -> Match | None:
|
|
727
788
|
with _client() as c:
|
|
728
|
-
r =
|
|
789
|
+
r = _get(
|
|
790
|
+
c, f"https://api.crossref.org/works/{doi}", params={"mailto": _mailto()}
|
|
791
|
+
)
|
|
729
792
|
if r.status_code != 200:
|
|
730
793
|
return None
|
|
731
794
|
data = r.json().get("message", {})
|
|
@@ -737,7 +800,9 @@ def crossref_by_doi(doi: str) -> Match | None:
|
|
|
737
800
|
if parts and parts[0]:
|
|
738
801
|
year = str(parts[0][0])
|
|
739
802
|
bibtex = ""
|
|
740
|
-
br =
|
|
803
|
+
br = _get(
|
|
804
|
+
c, f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex"
|
|
805
|
+
)
|
|
741
806
|
if br.status_code == 200:
|
|
742
807
|
bibtex = br.text
|
|
743
808
|
authors = [
|
|
@@ -815,12 +880,21 @@ def find_published(
|
|
|
815
880
|
# version (the common case) misses everywhere, and used to pay the *sum* of
|
|
816
881
|
# each source's latency; now the wall-clock is the slowest single source.
|
|
817
882
|
# The first verified hit by CASCADE priority still wins.
|
|
883
|
+
deadline = time.monotonic() + PUBLICATION_TIMEOUT
|
|
818
884
|
active = [(name, fn) for name, fn in CASCADE if name not in _DISABLED]
|
|
819
885
|
outcomes: dict[str, tuple] = {}
|
|
820
886
|
if active:
|
|
821
887
|
with ThreadPoolExecutor(max_workers=len(active)) as pool:
|
|
822
888
|
futures = {
|
|
823
|
-
pool.submit(
|
|
889
|
+
pool.submit(
|
|
890
|
+
_with_deadline,
|
|
891
|
+
deadline,
|
|
892
|
+
fn,
|
|
893
|
+
title,
|
|
894
|
+
year,
|
|
895
|
+
arxiv_id,
|
|
896
|
+
author_hint,
|
|
897
|
+
): name
|
|
824
898
|
for name, fn in active
|
|
825
899
|
}
|
|
826
900
|
for future in as_completed(futures):
|
|
@@ -858,9 +932,18 @@ def find_published(
|
|
|
858
932
|
# Exact-title search missed everywhere. Before concluding "no published
|
|
859
933
|
# version", try the title-drift fallback — camera-ready titles frequently
|
|
860
934
|
# differ from the arXiv ones, which is precisely the upgrade scenario.
|
|
861
|
-
|
|
935
|
+
# Only a clean exact DBLP miss justifies another query. A timeout or other
|
|
936
|
+
# failure has already spent its chance for this entry, and retrying the
|
|
937
|
+
# fuzzy form was doubling the worst-case interactive latency.
|
|
938
|
+
dblp_outcome = outcomes.get("dblp")
|
|
939
|
+
if (
|
|
940
|
+
author_hint
|
|
941
|
+
and dblp_outcome is not None
|
|
942
|
+
and dblp_outcome[0] == "miss"
|
|
943
|
+
and time.monotonic() < deadline
|
|
944
|
+
):
|
|
862
945
|
try:
|
|
863
|
-
m = try_dblp_fuzzy
|
|
946
|
+
m = _with_deadline(deadline, try_dblp_fuzzy, title, author_hint, year)
|
|
864
947
|
if m:
|
|
865
948
|
cache.put(cache_key, m.__dict__)
|
|
866
949
|
return m, "found"
|
|
@@ -920,7 +1003,7 @@ def fetch_web_page(url: str) -> WebPage:
|
|
|
920
1003
|
"""
|
|
921
1004
|
try:
|
|
922
1005
|
with _client(browser=True) as client:
|
|
923
|
-
response = client
|
|
1006
|
+
response = _get(client, url)
|
|
924
1007
|
response.raise_for_status()
|
|
925
1008
|
body = response.text[:400_000]
|
|
926
1009
|
except httpx.HTTPStatusError as e:
|
|
@@ -5,7 +5,12 @@ import bibcite.sources as sources
|
|
|
5
5
|
from bibcite import cache
|
|
6
6
|
from bibcite.bibfile import load_bib_file
|
|
7
7
|
from bibcite.cli import _upgrade_entries
|
|
8
|
-
from bibcite.sources import
|
|
8
|
+
from bibcite.sources import (
|
|
9
|
+
Match,
|
|
10
|
+
SourceUnavailable,
|
|
11
|
+
TransientSourceError,
|
|
12
|
+
find_published,
|
|
13
|
+
)
|
|
9
14
|
|
|
10
15
|
|
|
11
16
|
@pytest.fixture(autouse=True)
|
|
@@ -22,7 +27,7 @@ class _ReadErrorClient:
|
|
|
22
27
|
self.failures = failures
|
|
23
28
|
self.calls = 0
|
|
24
29
|
|
|
25
|
-
def get(self, url, params=None, headers=None):
|
|
30
|
+
def get(self, url, params=None, headers=None, timeout=None):
|
|
26
31
|
self.calls += 1
|
|
27
32
|
request = httpx.Request("GET", url, params=params, headers=headers)
|
|
28
33
|
if self.calls <= self.failures:
|
|
@@ -70,6 +75,41 @@ def test_dblp_read_failures_do_not_disable_later_batch_entries(monkeypatch):
|
|
|
70
75
|
assert second_match.venue == "TMLR"
|
|
71
76
|
|
|
72
77
|
|
|
78
|
+
def test_dblp_transport_failure_skips_the_fuzzy_retry(monkeypatch):
|
|
79
|
+
fuzzy_calls = 0
|
|
80
|
+
|
|
81
|
+
def dblp(*args):
|
|
82
|
+
raise TransientSourceError("simulated timeout")
|
|
83
|
+
|
|
84
|
+
def fuzzy(*args):
|
|
85
|
+
nonlocal fuzzy_calls
|
|
86
|
+
fuzzy_calls += 1
|
|
87
|
+
|
|
88
|
+
monkeypatch.setattr(
|
|
89
|
+
sources,
|
|
90
|
+
"CASCADE",
|
|
91
|
+
(
|
|
92
|
+
("dblp", dblp),
|
|
93
|
+
("crossref", lambda *args: None),
|
|
94
|
+
),
|
|
95
|
+
)
|
|
96
|
+
monkeypatch.setattr(sources, "try_dblp_fuzzy", fuzzy)
|
|
97
|
+
|
|
98
|
+
match, status = find_published("First paper", author_hint="doe")
|
|
99
|
+
|
|
100
|
+
assert (match, status) == (None, "incomplete")
|
|
101
|
+
assert fuzzy_calls == 0
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_requests_use_only_the_remaining_publication_budget(monkeypatch):
|
|
105
|
+
monkeypatch.setattr(sources.time, "monotonic", lambda: 7.0)
|
|
106
|
+
sources._REQUEST_DEADLINE.value = 10.0
|
|
107
|
+
try:
|
|
108
|
+
assert sources._request_timeout(8.0) == 3.0
|
|
109
|
+
finally:
|
|
110
|
+
del sources._REQUEST_DEADLINE.value
|
|
111
|
+
|
|
112
|
+
|
|
73
113
|
def test_upgrade_retries_dblp_after_previous_entry_read_failures(
|
|
74
114
|
tmp_path, monkeypatch
|
|
75
115
|
):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|