bibcite-cli 0.6.2__tar.gz → 0.6.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/.github/workflows/publish.yml +12 -4
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/PKG-INFO +6 -4
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/Readme.md +4 -2
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/pyproject.toml +1 -1
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/normalize.py +1 -2
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/sources.py +159 -76
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/venues.py +1 -1
- bibcite_cli-0.6.4/tests/test_no_google_scholar.py +69 -0
- bibcite_cli-0.6.4/tests/test_public_service.py +188 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_source_retries.py +42 -2
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_status_semantics.py +3 -3
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/uv.lock +1 -1
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/.github/workflows/ci.yml +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/.gitignore +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/LICENSE +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/assets/bibcite.svg +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/skills/bibcite/SKILL.md +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/__init__.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/bibfile.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/cli.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/src/bibcite/resolve.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_cli_status.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_round2.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_round3.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_venues.py +0 -0
- {bibcite_cli-0.6.2 → bibcite_cli-0.6.4}/tests/test_webpages.py +0 -0
|
@@ -4,6 +4,12 @@ on:
|
|
|
4
4
|
release:
|
|
5
5
|
types:
|
|
6
6
|
- published
|
|
7
|
+
workflow_dispatch:
|
|
8
|
+
inputs:
|
|
9
|
+
tag:
|
|
10
|
+
description: Release tag to publish
|
|
11
|
+
required: true
|
|
12
|
+
type: string
|
|
7
13
|
|
|
8
14
|
permissions:
|
|
9
15
|
contents: read
|
|
@@ -12,11 +18,13 @@ jobs:
|
|
|
12
18
|
build:
|
|
13
19
|
name: Build and verify distributions
|
|
14
20
|
runs-on: ubuntu-latest
|
|
21
|
+
env:
|
|
22
|
+
RELEASE_TAG: ${{ github.event.release.tag_name || inputs.tag }}
|
|
15
23
|
steps:
|
|
16
24
|
- name: Check out the release
|
|
17
25
|
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
|
|
18
26
|
with:
|
|
19
|
-
ref: ${{
|
|
27
|
+
ref: ${{ env.RELEASE_TAG }}
|
|
20
28
|
|
|
21
29
|
- name: Install uv and Python
|
|
22
30
|
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
|
@@ -26,8 +34,8 @@ jobs:
|
|
|
26
34
|
- name: Verify the release tag matches the package version
|
|
27
35
|
run: |
|
|
28
36
|
package_version="$(uv version --short)"
|
|
29
|
-
if [ "${
|
|
30
|
-
echo "Release tag ${
|
|
37
|
+
if [ "${RELEASE_TAG}" != "v${package_version}" ]; then
|
|
38
|
+
echo "Release tag ${RELEASE_TAG} does not match package version ${package_version}."
|
|
31
39
|
exit 1
|
|
32
40
|
fi
|
|
33
41
|
|
|
@@ -71,4 +79,4 @@ jobs:
|
|
|
71
79
|
path: dist/
|
|
72
80
|
|
|
73
81
|
- name: Publish distributions to PyPI
|
|
74
|
-
uses: pypa/gh-action-pypi-publish@
|
|
82
|
+
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.4
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -158,7 +158,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
|
|
|
158
158
|
|
|
159
159
|
## How resolution works
|
|
160
160
|
|
|
161
|
-
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar,
|
|
161
|
+
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
162
|
+
Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
|
|
162
163
|
A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
|
|
163
164
|
|
|
164
165
|
Successful published matches are cached at `~/.cache/bibcite/published.json`.
|
|
@@ -176,6 +177,7 @@ These optional environment variables improve source reliability:
|
|
|
176
177
|
| `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
|
|
177
178
|
| `S2_API_KEY` | Uses a private Semantic Scholar quota. |
|
|
178
179
|
| `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
|
|
180
|
+
| `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
|
|
179
181
|
| `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
|
|
180
182
|
| `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
|
|
181
183
|
|
|
@@ -211,7 +213,7 @@ uv tool install --editable .
|
|
|
211
213
|
|
|
212
214
|
## Acknowledgements
|
|
213
215
|
|
|
214
|
-
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar,
|
|
216
|
+
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
215
217
|
|
|
216
218
|
## License
|
|
217
219
|
|
|
@@ -145,7 +145,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
|
|
|
145
145
|
|
|
146
146
|
## How resolution works
|
|
147
147
|
|
|
148
|
-
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar,
|
|
148
|
+
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
149
|
+
Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
|
|
149
150
|
A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
|
|
150
151
|
|
|
151
152
|
Successful published matches are cached at `~/.cache/bibcite/published.json`.
|
|
@@ -163,6 +164,7 @@ These optional environment variables improve source reliability:
|
|
|
163
164
|
| `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
|
|
164
165
|
| `S2_API_KEY` | Uses a private Semantic Scholar quota. |
|
|
165
166
|
| `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
|
|
167
|
+
| `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
|
|
166
168
|
| `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
|
|
167
169
|
| `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
|
|
168
170
|
|
|
@@ -198,7 +200,7 @@ uv tool install --editable .
|
|
|
198
200
|
|
|
199
201
|
## Acknowledgements
|
|
200
202
|
|
|
201
|
-
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar,
|
|
203
|
+
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
202
204
|
|
|
203
205
|
## License
|
|
204
206
|
|
|
@@ -30,8 +30,7 @@ def mini_hash(s: str, replace: str = "") -> str:
|
|
|
30
30
|
"""PaperMemory's miniHash: lowercase, non-alphanumeric replaced.
|
|
31
31
|
|
|
32
32
|
When ``replace`` is non-empty, each non-word char maps to one replacement
|
|
33
|
-
char so string positions are preserved
|
|
34
|
-
parser).
|
|
33
|
+
char so string positions are preserved.
|
|
35
34
|
"""
|
|
36
35
|
if replace:
|
|
37
36
|
return re.sub(r"[^a-z0-9_]", replace, s.lower())
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""API clients for the publication-matching cascade.
|
|
2
2
|
|
|
3
3
|
Order and matching rules ported from PaperMemory's bibMatcher:
|
|
4
|
-
DBLP -> Semantic Scholar ->
|
|
4
|
+
DBLP -> Semantic Scholar -> CrossRef -> Unpaywall -> OpenAlex.
|
|
5
5
|
All matchers verify identity via normalized-title equality and reject
|
|
6
6
|
preprint venues (arXiv / CoRR / bioRxiv / ...).
|
|
7
7
|
"""
|
|
@@ -10,10 +10,12 @@ import html
|
|
|
10
10
|
import os
|
|
11
11
|
import re
|
|
12
12
|
import sys
|
|
13
|
+
import threading
|
|
13
14
|
import time
|
|
14
15
|
import xml.etree.ElementTree as ET
|
|
15
16
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
16
17
|
from dataclasses import dataclass, field
|
|
18
|
+
from urllib.parse import urlsplit
|
|
17
19
|
|
|
18
20
|
import httpx
|
|
19
21
|
|
|
@@ -30,6 +32,10 @@ BROWSER_UA = (
|
|
|
30
32
|
# never a false "not published". The arXiv metadata fetch sets its own longer
|
|
31
33
|
# timeout on the request itself, so this does not affect it.
|
|
32
34
|
TIMEOUT = 8.0
|
|
35
|
+
# Publication matching is enrichment on top of a valid arXiv citation. Keep
|
|
36
|
+
# the entire concurrent cascade, including its DBLP title-drift fallback,
|
|
37
|
+
# within an interactive budget instead of letting sequential retries add up.
|
|
38
|
+
PUBLICATION_TIMEOUT = 10.0
|
|
33
39
|
|
|
34
40
|
PREPRINT_VENUES = re.compile(r"arxiv|corr|biorxiv|medrxiv|chemrxiv|ssrn|preprint", re.I)
|
|
35
41
|
ARXIV_DOI = re.compile(r"^10\.48550/", re.I)
|
|
@@ -54,6 +60,112 @@ class TransientSourceError(SourceUnavailable):
|
|
|
54
60
|
process-wide circuit breaker for later batch entries."""
|
|
55
61
|
|
|
56
62
|
|
|
63
|
+
class PublicationTimeout(TransientSourceError):
|
|
64
|
+
"""The total publication-matching budget was exhausted."""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
_REQUEST_DEADLINE = threading.local()
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _request_timeout(cap: float = TIMEOUT) -> float:
|
|
71
|
+
deadline = getattr(_REQUEST_DEADLINE, "value", None)
|
|
72
|
+
if deadline is None:
|
|
73
|
+
return cap
|
|
74
|
+
remaining = deadline - time.monotonic()
|
|
75
|
+
if remaining <= 0:
|
|
76
|
+
raise PublicationTimeout("publication lookup timed out")
|
|
77
|
+
return max(0.001, min(cap, remaining))
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _get(
|
|
81
|
+
c: httpx.Client, url: str, *, timeout: float = TIMEOUT, **kwargs
|
|
82
|
+
) -> httpx.Response:
|
|
83
|
+
"""GET directly or route keyless supported APIs through the public service."""
|
|
84
|
+
public_url = os.environ.get("BIBCITE_PUBLIC_SERVICE_URL")
|
|
85
|
+
target = urlsplit(url)
|
|
86
|
+
providers = {
|
|
87
|
+
"api.openalex.org": "openalex",
|
|
88
|
+
"api.semanticscholar.org": "semanticscholar",
|
|
89
|
+
"api.crossref.org": "crossref",
|
|
90
|
+
}
|
|
91
|
+
provider = providers.get((target.hostname or "").lower())
|
|
92
|
+
personal_access = {
|
|
93
|
+
"openalex": bool(os.environ.get("OPENALEX_API_KEY")),
|
|
94
|
+
"semanticscholar": bool(
|
|
95
|
+
os.environ.get("S2_API_KEY")
|
|
96
|
+
or os.environ.get("SEMANTIC_SCHOLAR_API_KEY")
|
|
97
|
+
),
|
|
98
|
+
"crossref": bool(os.environ.get("BIBCITE_MAILTO")),
|
|
99
|
+
}
|
|
100
|
+
if not public_url or not provider or personal_access[provider]:
|
|
101
|
+
return c.get(url, timeout=_request_timeout(timeout), **kwargs)
|
|
102
|
+
|
|
103
|
+
service = urlsplit(public_url)
|
|
104
|
+
loopback = service.hostname in {"localhost", "127.0.0.1", "::1"}
|
|
105
|
+
if (
|
|
106
|
+
(service.scheme != "https" and not (service.scheme == "http" and loopback))
|
|
107
|
+
or not service.hostname
|
|
108
|
+
or service.username is not None
|
|
109
|
+
or service.password is not None
|
|
110
|
+
or service.query
|
|
111
|
+
or service.fragment
|
|
112
|
+
):
|
|
113
|
+
raise SourceUnavailable("BIBCITE_PUBLIC_SERVICE_URL is invalid")
|
|
114
|
+
|
|
115
|
+
params = kwargs.get("params") or {}
|
|
116
|
+
safe_params = {
|
|
117
|
+
str(key): str(value)
|
|
118
|
+
for key, value in params.items()
|
|
119
|
+
if str(key).lower() not in {"api_key", "mailto"}
|
|
120
|
+
}
|
|
121
|
+
try:
|
|
122
|
+
response = c.post(
|
|
123
|
+
public_url,
|
|
124
|
+
json={"provider": provider, "path": target.path, "params": safe_params},
|
|
125
|
+
timeout=_request_timeout(timeout),
|
|
126
|
+
follow_redirects=False,
|
|
127
|
+
)
|
|
128
|
+
except httpx.HTTPError as e:
|
|
129
|
+
raise SourceUnavailable(
|
|
130
|
+
f"public literature service unavailable ({type(e).__name__})"
|
|
131
|
+
) from e
|
|
132
|
+
if response.status_code == 404:
|
|
133
|
+
return response
|
|
134
|
+
if response.status_code == 429:
|
|
135
|
+
raise SourceUnavailable("public literature service rate-limited (429)")
|
|
136
|
+
if response.is_error:
|
|
137
|
+
raise SourceUnavailable(
|
|
138
|
+
f"public literature service error ({response.status_code})"
|
|
139
|
+
)
|
|
140
|
+
return response
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _sleep(delay: float):
|
|
144
|
+
"""Sleep for pacing/backoff without crossing the publication deadline."""
|
|
145
|
+
deadline = getattr(_REQUEST_DEADLINE, "value", None)
|
|
146
|
+
if deadline is None:
|
|
147
|
+
time.sleep(delay)
|
|
148
|
+
return
|
|
149
|
+
remaining = deadline - time.monotonic()
|
|
150
|
+
if remaining <= 0:
|
|
151
|
+
raise PublicationTimeout("publication lookup timed out")
|
|
152
|
+
time.sleep(min(delay, remaining))
|
|
153
|
+
if delay >= remaining:
|
|
154
|
+
raise PublicationTimeout("publication lookup timed out")
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _with_deadline(deadline: float, fn, *args):
|
|
158
|
+
previous = getattr(_REQUEST_DEADLINE, "value", None)
|
|
159
|
+
_REQUEST_DEADLINE.value = deadline
|
|
160
|
+
try:
|
|
161
|
+
return fn(*args)
|
|
162
|
+
finally:
|
|
163
|
+
if previous is None:
|
|
164
|
+
del _REQUEST_DEADLINE.value
|
|
165
|
+
else:
|
|
166
|
+
_REQUEST_DEADLINE.value = previous
|
|
167
|
+
|
|
168
|
+
|
|
57
169
|
def _client(browser: bool = False) -> httpx.Client:
|
|
58
170
|
return httpx.Client(
|
|
59
171
|
headers={"User-Agent": BROWSER_UA if browser else UA},
|
|
@@ -129,7 +241,8 @@ def arxiv_api_get(params: dict) -> httpx.Response:
|
|
|
129
241
|
time.sleep(3 * attempt)
|
|
130
242
|
try:
|
|
131
243
|
with _client() as c:
|
|
132
|
-
r =
|
|
244
|
+
r = _get(
|
|
245
|
+
c,
|
|
133
246
|
"https://export.arxiv.org/api/query",
|
|
134
247
|
params=params,
|
|
135
248
|
timeout=30.0,
|
|
@@ -198,13 +311,13 @@ def _paced_get(
|
|
|
198
311
|
for attempt in range(2):
|
|
199
312
|
wait = min_interval - (time.monotonic() - _LAST_REQUEST.get(source, 0.0))
|
|
200
313
|
if wait > 0:
|
|
201
|
-
|
|
314
|
+
_sleep(wait)
|
|
202
315
|
_LAST_REQUEST[source] = time.monotonic()
|
|
203
316
|
try:
|
|
204
|
-
r = c
|
|
317
|
+
r = _get(c, url, params=params, headers=headers)
|
|
205
318
|
except httpx.HTTPError as e: # Retry transport errors once before failing.
|
|
206
319
|
if attempt < 1:
|
|
207
|
-
|
|
320
|
+
_sleep(1)
|
|
208
321
|
continue
|
|
209
322
|
raise TransientSourceError(
|
|
210
323
|
f"{source} unreachable ({type(e).__name__})"
|
|
@@ -218,7 +331,7 @@ def _paced_get(
|
|
|
218
331
|
# skip this source for the rest of the run.
|
|
219
332
|
retry_after = int(r.headers.get("Retry-After") or 0)
|
|
220
333
|
if attempt < 1 and retry_after <= 2:
|
|
221
|
-
|
|
334
|
+
_sleep(max(retry_after, 1))
|
|
222
335
|
continue
|
|
223
336
|
raise SourceUnavailable(f"{source} rate-limited (429)")
|
|
224
337
|
return r
|
|
@@ -398,7 +511,7 @@ def arxiv_abs_metadata(arxiv_id: str) -> ArxivMeta | None:
|
|
|
398
511
|
"""Scrape the arxiv.org abs page's Highwire meta tags — the abs pages stay
|
|
399
512
|
up when the export API throttles."""
|
|
400
513
|
with _client(browser=True) as c:
|
|
401
|
-
r = c
|
|
514
|
+
r = _get(c, f"https://arxiv.org/abs/{arxiv_id}")
|
|
402
515
|
if r.status_code != 200:
|
|
403
516
|
return None
|
|
404
517
|
page = r.text
|
|
@@ -483,68 +596,14 @@ def try_semantic_scholar(
|
|
|
483
596
|
return None
|
|
484
597
|
|
|
485
598
|
|
|
486
|
-
# ---------------------------------------------------------------------------
|
|
487
|
-
# Google Scholar (port of PaperMemory's background fetchGSData)
|
|
488
|
-
# ---------------------------------------------------------------------------
|
|
489
|
-
|
|
490
|
-
def try_google_scholar(title: str) -> Match | None:
|
|
491
|
-
with _client(browser=True) as c:
|
|
492
|
-
r = c.get(
|
|
493
|
-
"https://scholar.google.com/scholar",
|
|
494
|
-
params={"q": title, "hl": "en"},
|
|
495
|
-
)
|
|
496
|
-
if r.status_code == 429 or "captcha" in r.text.lower()[:5000]:
|
|
497
|
-
raise SourceUnavailable("Google Scholar is blocking requests (captcha/429)")
|
|
498
|
-
r.raise_for_status()
|
|
499
|
-
parts = r.text.split("gs_res_ccl_mid")
|
|
500
|
-
if len(parts) < 2:
|
|
501
|
-
return None
|
|
502
|
-
page = parts[1]
|
|
503
|
-
# Each result title anchor looks like <a id="DATAID" href=...>Title</a>
|
|
504
|
-
# (the title may contain <b> highlights and HTML entities).
|
|
505
|
-
data_id = ""
|
|
506
|
-
for am in re.finditer(
|
|
507
|
-
r'<a[^>]*\bid="([\w-]{6,40})"[^>]*>(.*?)</a>', page, re.S
|
|
508
|
-
):
|
|
509
|
-
text = html.unescape(re.sub(r"<[^>]+>", "", am.group(2)))
|
|
510
|
-
if norm_title(text) == norm_title(title):
|
|
511
|
-
data_id = am.group(1)
|
|
512
|
-
break
|
|
513
|
-
if not data_id:
|
|
514
|
-
return None
|
|
515
|
-
cite_url = (
|
|
516
|
-
"https://scholar.google.com/scholar?q=info:"
|
|
517
|
-
f"{data_id}:scholar.google.com/&output=cite&scirp=0&hl=en"
|
|
518
|
-
)
|
|
519
|
-
cite_html = c.get(cite_url).text
|
|
520
|
-
bm = re.search(r'<a[^>]*href="([^">]+)"[^>]*>BibTex</a>', cite_html, re.I)
|
|
521
|
-
if not bm:
|
|
522
|
-
return None
|
|
523
|
-
bib_url = re.sub(r"\s+", "", bm.group(1).replace("&", "&"))
|
|
524
|
-
bibtex = c.get(bib_url).text
|
|
525
|
-
from .bibfile import parse_bibtex_entry # local import to avoid cycle
|
|
526
|
-
|
|
527
|
-
entry = parse_bibtex_entry(bibtex)
|
|
528
|
-
venue = entry.get("journal", "") or entry.get("booktitle", "")
|
|
529
|
-
if venue and not venue.lower().endswith("xiv") and "preprint" not in venue.lower():
|
|
530
|
-
_log(f"[googlescholar] match: {venue}")
|
|
531
|
-
return Match(
|
|
532
|
-
source="googlescholar",
|
|
533
|
-
venue=venue,
|
|
534
|
-
title=clean_title(entry.get("title", title)),
|
|
535
|
-
year=entry.get("year", ""),
|
|
536
|
-
bibtex=bibtex,
|
|
537
|
-
)
|
|
538
|
-
return None
|
|
539
|
-
|
|
540
|
-
|
|
541
599
|
# ---------------------------------------------------------------------------
|
|
542
600
|
# CrossRef
|
|
543
601
|
# ---------------------------------------------------------------------------
|
|
544
602
|
|
|
545
603
|
def try_crossref(title: str) -> Match | None:
|
|
546
604
|
with _client() as c:
|
|
547
|
-
r =
|
|
605
|
+
r = _get(
|
|
606
|
+
c,
|
|
548
607
|
"https://api.crossref.org/works",
|
|
549
608
|
params={
|
|
550
609
|
"rows": 3,
|
|
@@ -583,8 +642,9 @@ def try_crossref(title: str) -> Match | None:
|
|
|
583
642
|
year = str(parts[0][0])
|
|
584
643
|
bibtex = ""
|
|
585
644
|
if doi:
|
|
586
|
-
br =
|
|
587
|
-
|
|
645
|
+
br = _get(
|
|
646
|
+
c,
|
|
647
|
+
f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex",
|
|
588
648
|
)
|
|
589
649
|
if br.status_code == 200:
|
|
590
650
|
bibtex = br.text
|
|
@@ -606,7 +666,8 @@ def try_crossref(title: str) -> Match | None:
|
|
|
606
666
|
|
|
607
667
|
def try_unpaywall(title: str) -> Match | None:
|
|
608
668
|
with _client() as c:
|
|
609
|
-
r =
|
|
669
|
+
r = _get(
|
|
670
|
+
c,
|
|
610
671
|
"https://api.unpaywall.org/v2/search",
|
|
611
672
|
params={"query": title, "is_oa": "true", "email": _mailto()},
|
|
612
673
|
)
|
|
@@ -653,7 +714,8 @@ def try_unpaywall(title: str) -> Match | None:
|
|
|
653
714
|
def openalex_search(title: str) -> dict | None:
|
|
654
715
|
"""OpenAlex work with an exactly-matching normalized title, or None."""
|
|
655
716
|
with _client() as c:
|
|
656
|
-
r =
|
|
717
|
+
r = _get(
|
|
718
|
+
c,
|
|
657
719
|
"https://api.openalex.org/works",
|
|
658
720
|
params=_openalex_params({"search": title, "per-page": 5}),
|
|
659
721
|
)
|
|
@@ -725,7 +787,9 @@ def try_openalex(title: str) -> Match | None:
|
|
|
725
787
|
|
|
726
788
|
def crossref_by_doi(doi: str) -> Match | None:
|
|
727
789
|
with _client() as c:
|
|
728
|
-
r =
|
|
790
|
+
r = _get(
|
|
791
|
+
c, f"https://api.crossref.org/works/{doi}", params={"mailto": _mailto()}
|
|
792
|
+
)
|
|
729
793
|
if r.status_code != 200:
|
|
730
794
|
return None
|
|
731
795
|
data = r.json().get("message", {})
|
|
@@ -737,7 +801,9 @@ def crossref_by_doi(doi: str) -> Match | None:
|
|
|
737
801
|
if parts and parts[0]:
|
|
738
802
|
year = str(parts[0][0])
|
|
739
803
|
bibtex = ""
|
|
740
|
-
br =
|
|
804
|
+
br = _get(
|
|
805
|
+
c, f"https://api.crossref.org/works/{doi}/transform/application/x-bibtex"
|
|
806
|
+
)
|
|
741
807
|
if br.status_code == 200:
|
|
742
808
|
bibtex = br.text
|
|
743
809
|
authors = [
|
|
@@ -762,7 +828,6 @@ def crossref_by_doi(doi: str) -> Match | None:
|
|
|
762
828
|
CASCADE = (
|
|
763
829
|
("dblp", lambda t, y, a, au: try_dblp(t, au)),
|
|
764
830
|
("semanticscholar", lambda t, y, a, au: try_semantic_scholar(t, y, a)),
|
|
765
|
-
("googlescholar", lambda t, y, a, au: try_google_scholar(t)),
|
|
766
831
|
("crossref", lambda t, y, a, au: try_crossref(t)),
|
|
767
832
|
("unpaywall", lambda t, y, a, au: try_unpaywall(t)),
|
|
768
833
|
("openalex", lambda t, y, a, au: try_openalex(t)),
|
|
@@ -775,8 +840,8 @@ CASCADE = (
|
|
|
775
840
|
_DISABLED: dict[str, str] = {}
|
|
776
841
|
|
|
777
842
|
# Only these sources are authoritative enough that losing one taints a miss
|
|
778
|
-
# into "incomplete".
|
|
779
|
-
#
|
|
843
|
+
# into "incomplete". Unpaywall flakiness is routine and must not stop
|
|
844
|
+
# "not_found" from ever being trustworthy.
|
|
780
845
|
# Override with BIBCITE_CORE_SOURCES="dblp,semanticscholar" if one of these
|
|
781
846
|
# is down for days and keeps every verdict incomplete.
|
|
782
847
|
CORE_SOURCES = frozenset(
|
|
@@ -815,12 +880,21 @@ def find_published(
|
|
|
815
880
|
# version (the common case) misses everywhere, and used to pay the *sum* of
|
|
816
881
|
# each source's latency; now the wall-clock is the slowest single source.
|
|
817
882
|
# The first verified hit by CASCADE priority still wins.
|
|
883
|
+
deadline = time.monotonic() + PUBLICATION_TIMEOUT
|
|
818
884
|
active = [(name, fn) for name, fn in CASCADE if name not in _DISABLED]
|
|
819
885
|
outcomes: dict[str, tuple] = {}
|
|
820
886
|
if active:
|
|
821
887
|
with ThreadPoolExecutor(max_workers=len(active)) as pool:
|
|
822
888
|
futures = {
|
|
823
|
-
pool.submit(
|
|
889
|
+
pool.submit(
|
|
890
|
+
_with_deadline,
|
|
891
|
+
deadline,
|
|
892
|
+
fn,
|
|
893
|
+
title,
|
|
894
|
+
year,
|
|
895
|
+
arxiv_id,
|
|
896
|
+
author_hint,
|
|
897
|
+
): name
|
|
824
898
|
for name, fn in active
|
|
825
899
|
}
|
|
826
900
|
for future in as_completed(futures):
|
|
@@ -858,9 +932,18 @@ def find_published(
|
|
|
858
932
|
# Exact-title search missed everywhere. Before concluding "no published
|
|
859
933
|
# version", try the title-drift fallback — camera-ready titles frequently
|
|
860
934
|
# differ from the arXiv ones, which is precisely the upgrade scenario.
|
|
861
|
-
|
|
935
|
+
# Only a clean exact DBLP miss justifies another query. A timeout or other
|
|
936
|
+
# failure has already spent its chance for this entry, and retrying the
|
|
937
|
+
# fuzzy form was doubling the worst-case interactive latency.
|
|
938
|
+
dblp_outcome = outcomes.get("dblp")
|
|
939
|
+
if (
|
|
940
|
+
author_hint
|
|
941
|
+
and dblp_outcome is not None
|
|
942
|
+
and dblp_outcome[0] == "miss"
|
|
943
|
+
and time.monotonic() < deadline
|
|
944
|
+
):
|
|
862
945
|
try:
|
|
863
|
-
m = try_dblp_fuzzy
|
|
946
|
+
m = _with_deadline(deadline, try_dblp_fuzzy, title, author_hint, year)
|
|
864
947
|
if m:
|
|
865
948
|
cache.put(cache_key, m.__dict__)
|
|
866
949
|
return m, "found"
|
|
@@ -920,7 +1003,7 @@ def fetch_web_page(url: str) -> WebPage:
|
|
|
920
1003
|
"""
|
|
921
1004
|
try:
|
|
922
1005
|
with _client(browser=True) as client:
|
|
923
|
-
response = client
|
|
1006
|
+
response = _get(client, url)
|
|
924
1007
|
response.raise_for_status()
|
|
925
1008
|
body = response.text[:400_000]
|
|
926
1009
|
except httpx.HTTPStatusError as e:
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Parses the vendored ``data/strings.bib`` @string table (journals /
|
|
4
4
|
conferences / workshops) and maps venue strings returned by DBLP, Semantic
|
|
5
|
-
Scholar,
|
|
5
|
+
Scholar, CrossRef, Unpaywall, etc. onto the canonical names.
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
8
|
import re
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Exercise the CLI publication cascade without contacting external services."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import json
|
|
5
|
+
|
|
6
|
+
import httpx
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from bibcite import cache, cli, sources
|
|
10
|
+
|
|
11
|
+
resolver = importlib.import_module("bibcite.resolve")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@pytest.mark.parametrize("operation", ["get", "add", "upgrade"])
|
|
15
|
+
def test_cli_never_queries_google_scholar(operation, monkeypatch, tmp_path, capsys):
|
|
16
|
+
monkeypatch.setattr(cache, "DISABLED", True)
|
|
17
|
+
monkeypatch.setattr(sources, "_DISABLED", {})
|
|
18
|
+
monkeypatch.setattr(
|
|
19
|
+
resolver,
|
|
20
|
+
"arxiv_metadata",
|
|
21
|
+
lambda _: sources.ArxivMeta(
|
|
22
|
+
"1706.03762",
|
|
23
|
+
"Attention Is All You Need",
|
|
24
|
+
["Ashish Vaswani"],
|
|
25
|
+
"2017",
|
|
26
|
+
"https://arxiv.org/abs/1706.03762",
|
|
27
|
+
),
|
|
28
|
+
)
|
|
29
|
+
visited = []
|
|
30
|
+
for name in ("dblp", "semantic_scholar", "crossref", "unpaywall", "openalex"):
|
|
31
|
+
|
|
32
|
+
def miss(*args, source=name):
|
|
33
|
+
visited.append(source)
|
|
34
|
+
return None
|
|
35
|
+
|
|
36
|
+
monkeypatch.setattr(sources, f"try_{name}", miss)
|
|
37
|
+
monkeypatch.setattr(sources, "try_dblp_fuzzy", lambda *args: None)
|
|
38
|
+
requests = []
|
|
39
|
+
|
|
40
|
+
def unexpected_request(client, url, **kwargs):
|
|
41
|
+
requests.append(url)
|
|
42
|
+
return httpx.Response(429, request=httpx.Request("GET", url))
|
|
43
|
+
|
|
44
|
+
monkeypatch.setattr(sources, "_get", unexpected_request)
|
|
45
|
+
path = tmp_path / "references.bib"
|
|
46
|
+
if operation == "get":
|
|
47
|
+
args = ["get", "--json", "1706.03762"]
|
|
48
|
+
elif operation == "add":
|
|
49
|
+
args = ["add", "--no-tidy", str(path), "1706.03762"]
|
|
50
|
+
else:
|
|
51
|
+
path.write_text("""@article{vaswani2017attention,
|
|
52
|
+
title = {Attention Is All You Need},
|
|
53
|
+
author = {Ashish Vaswani},
|
|
54
|
+
year = {2017},
|
|
55
|
+
journal = {arXiv preprint arXiv:1706.03762},
|
|
56
|
+
eprint = {1706.03762}
|
|
57
|
+
}
|
|
58
|
+
""")
|
|
59
|
+
args = ["upgrade", str(path), "--dry-run"]
|
|
60
|
+
cli.main(args)
|
|
61
|
+
json.loads(capsys.readouterr().out)
|
|
62
|
+
assert set(visited) == {
|
|
63
|
+
"dblp",
|
|
64
|
+
"semantic_scholar",
|
|
65
|
+
"crossref",
|
|
66
|
+
"unpaywall",
|
|
67
|
+
"openalex",
|
|
68
|
+
}
|
|
69
|
+
assert requests == []
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
import httpx
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from bibcite import sources
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@pytest.fixture(autouse=True)
|
|
10
|
+
def clean_environment(monkeypatch):
|
|
11
|
+
for name in (
|
|
12
|
+
"BIBCITE_PUBLIC_SERVICE_URL",
|
|
13
|
+
"BIBCITE_MAILTO",
|
|
14
|
+
"OPENALEX_API_KEY",
|
|
15
|
+
"S2_API_KEY",
|
|
16
|
+
"SEMANTIC_SCHOLAR_API_KEY",
|
|
17
|
+
):
|
|
18
|
+
monkeypatch.delenv(name, raising=False)
|
|
19
|
+
monkeypatch.setattr(sources, "_LAST_REQUEST", {})
|
|
20
|
+
monkeypatch.setattr(sources.time, "sleep", lambda _: None)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _client(handler):
|
|
24
|
+
return httpx.Client(transport=httpx.MockTransport(handler))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@pytest.mark.parametrize(
|
|
28
|
+
("url", "provider", "path"),
|
|
29
|
+
(
|
|
30
|
+
("https://api.openalex.org/works", "openalex", "/works"),
|
|
31
|
+
(
|
|
32
|
+
"https://api.semanticscholar.org/graph/v1/paper/search",
|
|
33
|
+
"semanticscholar",
|
|
34
|
+
"/graph/v1/paper/search",
|
|
35
|
+
),
|
|
36
|
+
(
|
|
37
|
+
"https://api.crossref.org/works/10.1234/example/transform/application/x-bibtex",
|
|
38
|
+
"crossref",
|
|
39
|
+
"/works/10.1234/example/transform/application/x-bibtex",
|
|
40
|
+
),
|
|
41
|
+
),
|
|
42
|
+
)
|
|
43
|
+
def test_keyless_supported_requests_route_to_public_service(
|
|
44
|
+
url, provider, path, monkeypatch
|
|
45
|
+
):
|
|
46
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
47
|
+
requests = []
|
|
48
|
+
|
|
49
|
+
def handler(request):
|
|
50
|
+
requests.append(request)
|
|
51
|
+
return httpx.Response(200, json={"unchanged": True})
|
|
52
|
+
|
|
53
|
+
with _client(handler) as client:
|
|
54
|
+
response = sources._get(
|
|
55
|
+
client,
|
|
56
|
+
url,
|
|
57
|
+
params={
|
|
58
|
+
"query": "paper",
|
|
59
|
+
"limit": 5,
|
|
60
|
+
"api_key": "secret",
|
|
61
|
+
"mailto": "secret@example.com",
|
|
62
|
+
},
|
|
63
|
+
headers={"x-api-key": "secret"},
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
assert response.json() == {"unchanged": True}
|
|
67
|
+
assert len(requests) == 1
|
|
68
|
+
request = requests[0]
|
|
69
|
+
assert request.method == "POST"
|
|
70
|
+
assert request.url == "https://literature.test/v1/query"
|
|
71
|
+
assert "x-api-key" not in request.headers
|
|
72
|
+
payload = json.loads(request.content)
|
|
73
|
+
assert payload == {
|
|
74
|
+
"provider": provider,
|
|
75
|
+
"path": path,
|
|
76
|
+
"params": {"query": "paper", "limit": "5"},
|
|
77
|
+
}
|
|
78
|
+
assert "secret" not in request.content.decode()
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@pytest.mark.parametrize(
|
|
82
|
+
("variable", "url", "params"),
|
|
83
|
+
(
|
|
84
|
+
("OPENALEX_API_KEY", "https://api.openalex.org/works", {"api_key": "mine"}),
|
|
85
|
+
("S2_API_KEY", "https://api.semanticscholar.org/graph/v1/paper/search", {}),
|
|
86
|
+
(
|
|
87
|
+
"BIBCITE_MAILTO",
|
|
88
|
+
"https://api.crossref.org/works",
|
|
89
|
+
{"mailto": "me@example.com"},
|
|
90
|
+
),
|
|
91
|
+
),
|
|
92
|
+
)
|
|
93
|
+
def test_personal_access_overrides_public_service(variable, url, params, monkeypatch):
|
|
94
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
95
|
+
monkeypatch.setenv(variable, "mine")
|
|
96
|
+
requests = []
|
|
97
|
+
|
|
98
|
+
def handler(request):
|
|
99
|
+
requests.append(request)
|
|
100
|
+
return httpx.Response(200, json={})
|
|
101
|
+
|
|
102
|
+
with _client(handler) as client:
|
|
103
|
+
sources._get(client, url, params=params)
|
|
104
|
+
|
|
105
|
+
assert len(requests) == 1
|
|
106
|
+
assert requests[0].method == "GET"
|
|
107
|
+
assert requests[0].url.host != "literature.test"
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def test_without_public_service_requests_remain_direct():
|
|
111
|
+
requests = []
|
|
112
|
+
|
|
113
|
+
def handler(request):
|
|
114
|
+
requests.append(request)
|
|
115
|
+
return httpx.Response(200, json={})
|
|
116
|
+
|
|
117
|
+
with _client(handler) as client:
|
|
118
|
+
sources._get(client, "https://api.openalex.org/works", params={"search": "x"})
|
|
119
|
+
|
|
120
|
+
assert len(requests) == 1
|
|
121
|
+
assert requests[0].method == "GET"
|
|
122
|
+
assert requests[0].url.host == "api.openalex.org"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@pytest.mark.parametrize("status", [429, 500])
|
|
126
|
+
def test_public_service_http_failure_is_not_retried(status, monkeypatch):
|
|
127
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
128
|
+
calls = 0
|
|
129
|
+
|
|
130
|
+
def handler(request):
|
|
131
|
+
nonlocal calls
|
|
132
|
+
calls += 1
|
|
133
|
+
return httpx.Response(status)
|
|
134
|
+
|
|
135
|
+
with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
|
|
136
|
+
sources._paced_get(
|
|
137
|
+
client,
|
|
138
|
+
"https://api.semanticscholar.org/graph/v1/paper/search",
|
|
139
|
+
"semanticscholar",
|
|
140
|
+
0,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
assert calls == 1
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def test_public_service_timeout_is_not_retried(monkeypatch):
|
|
147
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
148
|
+
calls = 0
|
|
149
|
+
|
|
150
|
+
def handler(request):
|
|
151
|
+
nonlocal calls
|
|
152
|
+
calls += 1
|
|
153
|
+
raise httpx.ReadTimeout("timed out", request=request)
|
|
154
|
+
|
|
155
|
+
with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
|
|
156
|
+
sources._paced_get(
|
|
157
|
+
client,
|
|
158
|
+
"https://api.semanticscholar.org/graph/v1/paper/search",
|
|
159
|
+
"semanticscholar",
|
|
160
|
+
0,
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
assert calls == 1
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_public_service_404_is_preserved(monkeypatch):
|
|
167
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
168
|
+
|
|
169
|
+
with _client(lambda request: httpx.Response(404)) as client:
|
|
170
|
+
response = sources._get(client, "https://api.openalex.org/works/W1")
|
|
171
|
+
|
|
172
|
+
assert response.status_code == 404
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@pytest.mark.parametrize(
|
|
176
|
+
"value",
|
|
177
|
+
(
|
|
178
|
+
"http://literature.test/v1/query",
|
|
179
|
+
"https://user:pass@literature.test/v1/query",
|
|
180
|
+
"https://literature.test/v1/query?secret=x",
|
|
181
|
+
"https://literature.test/v1/query#fragment",
|
|
182
|
+
),
|
|
183
|
+
)
|
|
184
|
+
def test_invalid_public_service_url_is_rejected(value, monkeypatch):
|
|
185
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", value)
|
|
186
|
+
with _client(lambda request: httpx.Response(200)) as client:
|
|
187
|
+
with pytest.raises(sources.SourceUnavailable):
|
|
188
|
+
sources._get(client, "https://api.openalex.org/works")
|
|
@@ -5,7 +5,12 @@ import bibcite.sources as sources
|
|
|
5
5
|
from bibcite import cache
|
|
6
6
|
from bibcite.bibfile import load_bib_file
|
|
7
7
|
from bibcite.cli import _upgrade_entries
|
|
8
|
-
from bibcite.sources import
|
|
8
|
+
from bibcite.sources import (
|
|
9
|
+
Match,
|
|
10
|
+
SourceUnavailable,
|
|
11
|
+
TransientSourceError,
|
|
12
|
+
find_published,
|
|
13
|
+
)
|
|
9
14
|
|
|
10
15
|
|
|
11
16
|
@pytest.fixture(autouse=True)
|
|
@@ -22,7 +27,7 @@ class _ReadErrorClient:
|
|
|
22
27
|
self.failures = failures
|
|
23
28
|
self.calls = 0
|
|
24
29
|
|
|
25
|
-
def get(self, url, params=None, headers=None):
|
|
30
|
+
def get(self, url, params=None, headers=None, timeout=None):
|
|
26
31
|
self.calls += 1
|
|
27
32
|
request = httpx.Request("GET", url, params=params, headers=headers)
|
|
28
33
|
if self.calls <= self.failures:
|
|
@@ -70,6 +75,41 @@ def test_dblp_read_failures_do_not_disable_later_batch_entries(monkeypatch):
|
|
|
70
75
|
assert second_match.venue == "TMLR"
|
|
71
76
|
|
|
72
77
|
|
|
78
|
+
def test_dblp_transport_failure_skips_the_fuzzy_retry(monkeypatch):
|
|
79
|
+
fuzzy_calls = 0
|
|
80
|
+
|
|
81
|
+
def dblp(*args):
|
|
82
|
+
raise TransientSourceError("simulated timeout")
|
|
83
|
+
|
|
84
|
+
def fuzzy(*args):
|
|
85
|
+
nonlocal fuzzy_calls
|
|
86
|
+
fuzzy_calls += 1
|
|
87
|
+
|
|
88
|
+
monkeypatch.setattr(
|
|
89
|
+
sources,
|
|
90
|
+
"CASCADE",
|
|
91
|
+
(
|
|
92
|
+
("dblp", dblp),
|
|
93
|
+
("crossref", lambda *args: None),
|
|
94
|
+
),
|
|
95
|
+
)
|
|
96
|
+
monkeypatch.setattr(sources, "try_dblp_fuzzy", fuzzy)
|
|
97
|
+
|
|
98
|
+
match, status = find_published("First paper", author_hint="doe")
|
|
99
|
+
|
|
100
|
+
assert (match, status) == (None, "incomplete")
|
|
101
|
+
assert fuzzy_calls == 0
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_requests_use_only_the_remaining_publication_budget(monkeypatch):
|
|
105
|
+
monkeypatch.setattr(sources.time, "monotonic", lambda: 7.0)
|
|
106
|
+
sources._REQUEST_DEADLINE.value = 10.0
|
|
107
|
+
try:
|
|
108
|
+
assert sources._request_timeout(8.0) == 3.0
|
|
109
|
+
finally:
|
|
110
|
+
del sources._REQUEST_DEADLINE.value
|
|
111
|
+
|
|
112
|
+
|
|
73
113
|
def test_upgrade_retries_dblp_after_previous_entry_read_failures(
|
|
74
114
|
tmp_path, monkeypatch
|
|
75
115
|
):
|
|
@@ -43,7 +43,7 @@ def test_core_source_429_taints_verdict(monkeypatch):
|
|
|
43
43
|
monkeypatch.setattr(
|
|
44
44
|
sources,
|
|
45
45
|
"CASCADE",
|
|
46
|
-
_cascade(dblp="raise",
|
|
46
|
+
_cascade(dblp="raise", unpaywall=None, crossref=None),
|
|
47
47
|
)
|
|
48
48
|
match, status = find_published("Some Title", author_hint="smith")
|
|
49
49
|
assert (match, status) == (None, "incomplete")
|
|
@@ -60,11 +60,11 @@ def test_previously_disabled_core_source_taints_next_queries(monkeypatch):
|
|
|
60
60
|
|
|
61
61
|
|
|
62
62
|
def test_noncore_outage_does_not_taint(monkeypatch):
|
|
63
|
-
#
|
|
63
|
+
# Unpaywall outages are routine; a miss stays trustworthy.
|
|
64
64
|
monkeypatch.setattr(
|
|
65
65
|
sources,
|
|
66
66
|
"CASCADE",
|
|
67
|
-
_cascade(dblp=None,
|
|
67
|
+
_cascade(dblp=None, unpaywall="raise", crossref=None),
|
|
68
68
|
)
|
|
69
69
|
match, status = find_published("Some Title", author_hint="smith")
|
|
70
70
|
assert (match, status) == (None, "not_found")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|