bibcite-cli 0.6.3__tar.gz → 0.6.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/.github/workflows/publish.yml +12 -4
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/PKG-INFO +5 -3
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/Readme.md +4 -2
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/pyproject.toml +1 -1
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/normalize.py +1 -2
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/sources.py +62 -62
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/venues.py +1 -1
- bibcite_cli-0.6.4/tests/test_no_google_scholar.py +69 -0
- bibcite_cli-0.6.4/tests/test_public_service.py +188 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_status_semantics.py +3 -3
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/uv.lock +1 -1
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/.github/workflows/ci.yml +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/.gitignore +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/LICENSE +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/assets/bibcite.svg +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/skills/bibcite/SKILL.md +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/__init__.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/bibfile.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/cli.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/src/bibcite/resolve.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_cli_status.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_round2.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_round3.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_source_retries.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_venues.py +0 -0
- {bibcite_cli-0.6.3 → bibcite_cli-0.6.4}/tests/test_webpages.py +0 -0
|
@@ -4,6 +4,12 @@ on:
|
|
|
4
4
|
release:
|
|
5
5
|
types:
|
|
6
6
|
- published
|
|
7
|
+
workflow_dispatch:
|
|
8
|
+
inputs:
|
|
9
|
+
tag:
|
|
10
|
+
description: Release tag to publish
|
|
11
|
+
required: true
|
|
12
|
+
type: string
|
|
7
13
|
|
|
8
14
|
permissions:
|
|
9
15
|
contents: read
|
|
@@ -12,11 +18,13 @@ jobs:
|
|
|
12
18
|
build:
|
|
13
19
|
name: Build and verify distributions
|
|
14
20
|
runs-on: ubuntu-latest
|
|
21
|
+
env:
|
|
22
|
+
RELEASE_TAG: ${{ github.event.release.tag_name || inputs.tag }}
|
|
15
23
|
steps:
|
|
16
24
|
- name: Check out the release
|
|
17
25
|
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
|
|
18
26
|
with:
|
|
19
|
-
ref: ${{
|
|
27
|
+
ref: ${{ env.RELEASE_TAG }}
|
|
20
28
|
|
|
21
29
|
- name: Install uv and Python
|
|
22
30
|
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
|
@@ -26,8 +34,8 @@ jobs:
|
|
|
26
34
|
- name: Verify the release tag matches the package version
|
|
27
35
|
run: |
|
|
28
36
|
package_version="$(uv version --short)"
|
|
29
|
-
if [ "${
|
|
30
|
-
echo "Release tag ${
|
|
37
|
+
if [ "${RELEASE_TAG}" != "v${package_version}" ]; then
|
|
38
|
+
echo "Release tag ${RELEASE_TAG} does not match package version ${package_version}."
|
|
31
39
|
exit 1
|
|
32
40
|
fi
|
|
33
41
|
|
|
@@ -71,4 +79,4 @@ jobs:
|
|
|
71
79
|
path: dist/
|
|
72
80
|
|
|
73
81
|
- name: Publish distributions to PyPI
|
|
74
|
-
uses: pypa/gh-action-pypi-publish@
|
|
82
|
+
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.4
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -158,7 +158,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
|
|
|
158
158
|
|
|
159
159
|
## How resolution works
|
|
160
160
|
|
|
161
|
-
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar,
|
|
161
|
+
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
162
|
+
Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
|
|
162
163
|
A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
|
|
163
164
|
|
|
164
165
|
Successful published matches are cached at `~/.cache/bibcite/published.json`.
|
|
@@ -176,6 +177,7 @@ These optional environment variables improve source reliability:
|
|
|
176
177
|
| `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
|
|
177
178
|
| `S2_API_KEY` | Uses a private Semantic Scholar quota. |
|
|
178
179
|
| `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
|
|
180
|
+
| `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
|
|
179
181
|
| `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
|
|
180
182
|
| `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
|
|
181
183
|
|
|
@@ -211,7 +213,7 @@ uv tool install --editable .
|
|
|
211
213
|
|
|
212
214
|
## Acknowledgements
|
|
213
215
|
|
|
214
|
-
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar,
|
|
216
|
+
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
215
217
|
|
|
216
218
|
## License
|
|
217
219
|
|
|
@@ -145,7 +145,8 @@ Mark a confirmed preprint-only entry with `pubstate = {preprint}` if you want `c
|
|
|
145
145
|
|
|
146
146
|
## How resolution works
|
|
147
147
|
|
|
148
|
-
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar,
|
|
148
|
+
For arXiv IDs and titles, `bibcite` collects paper metadata and checks publication sources in a cascade derived from [PaperMemory](https://github.com/vict0rsch/PaperMemory): DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
149
|
+
Google Scholar scraping is excluded so CAPTCHA challenges cannot delay publication lookup.
|
|
149
150
|
A published match must have the same normalized title or pass a guarded title-drift check, have a plausible publication year, and name a non-preprint venue.
|
|
150
151
|
|
|
151
152
|
Successful published matches are cached at `~/.cache/bibcite/published.json`.
|
|
@@ -163,6 +164,7 @@ These optional environment variables improve source reliability:
|
|
|
163
164
|
| `OPENALEX_API_KEY` | Uses your OpenAlex quota instead of the anonymous shared pool. |
|
|
164
165
|
| `S2_API_KEY` | Uses a private Semantic Scholar quota. |
|
|
165
166
|
| `BIBCITE_MAILTO` | Sends your contact email to the Crossref, OpenAlex, and Unpaywall polite pools. |
|
|
167
|
+
| `BIBCITE_PUBLIC_SERVICE_URL` | Routes keyless OpenAlex, Semantic Scholar, and Crossref requests through an HTTPS `/v1/query` service supplied by the embedding application. |
|
|
166
168
|
| `BIBCITE_CORE_SOURCES` | Overrides the sources required for a trustworthy publication check. |
|
|
167
169
|
| `BIBCITE_NO_CACHE=1` | Disables the local publication cache. |
|
|
168
170
|
|
|
@@ -198,7 +200,7 @@ uv tool install --editable .
|
|
|
198
200
|
|
|
199
201
|
## Acknowledgements
|
|
200
202
|
|
|
201
|
-
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar,
|
|
203
|
+
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
202
204
|
|
|
203
205
|
## License
|
|
204
206
|
|
|
@@ -30,8 +30,7 @@ def mini_hash(s: str, replace: str = "") -> str:
|
|
|
30
30
|
"""PaperMemory's miniHash: lowercase, non-alphanumeric replaced.
|
|
31
31
|
|
|
32
32
|
When ``replace`` is non-empty, each non-word char maps to one replacement
|
|
33
|
-
char so string positions are preserved
|
|
34
|
-
parser).
|
|
33
|
+
char so string positions are preserved.
|
|
35
34
|
"""
|
|
36
35
|
if replace:
|
|
37
36
|
return re.sub(r"[^a-z0-9_]", replace, s.lower())
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""API clients for the publication-matching cascade.
|
|
2
2
|
|
|
3
3
|
Order and matching rules ported from PaperMemory's bibMatcher:
|
|
4
|
-
DBLP -> Semantic Scholar ->
|
|
4
|
+
DBLP -> Semantic Scholar -> CrossRef -> Unpaywall -> OpenAlex.
|
|
5
5
|
All matchers verify identity via normalized-title equality and reject
|
|
6
6
|
preprint venues (arXiv / CoRR / bioRxiv / ...).
|
|
7
7
|
"""
|
|
@@ -15,6 +15,7 @@ import time
|
|
|
15
15
|
import xml.etree.ElementTree as ET
|
|
16
16
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
17
17
|
from dataclasses import dataclass, field
|
|
18
|
+
from urllib.parse import urlsplit
|
|
18
19
|
|
|
19
20
|
import httpx
|
|
20
21
|
|
|
@@ -79,8 +80,64 @@ def _request_timeout(cap: float = TIMEOUT) -> float:
|
|
|
79
80
|
def _get(
|
|
80
81
|
c: httpx.Client, url: str, *, timeout: float = TIMEOUT, **kwargs
|
|
81
82
|
) -> httpx.Response:
|
|
82
|
-
"""GET
|
|
83
|
-
|
|
83
|
+
"""GET directly or route keyless supported APIs through the public service."""
|
|
84
|
+
public_url = os.environ.get("BIBCITE_PUBLIC_SERVICE_URL")
|
|
85
|
+
target = urlsplit(url)
|
|
86
|
+
providers = {
|
|
87
|
+
"api.openalex.org": "openalex",
|
|
88
|
+
"api.semanticscholar.org": "semanticscholar",
|
|
89
|
+
"api.crossref.org": "crossref",
|
|
90
|
+
}
|
|
91
|
+
provider = providers.get((target.hostname or "").lower())
|
|
92
|
+
personal_access = {
|
|
93
|
+
"openalex": bool(os.environ.get("OPENALEX_API_KEY")),
|
|
94
|
+
"semanticscholar": bool(
|
|
95
|
+
os.environ.get("S2_API_KEY")
|
|
96
|
+
or os.environ.get("SEMANTIC_SCHOLAR_API_KEY")
|
|
97
|
+
),
|
|
98
|
+
"crossref": bool(os.environ.get("BIBCITE_MAILTO")),
|
|
99
|
+
}
|
|
100
|
+
if not public_url or not provider or personal_access[provider]:
|
|
101
|
+
return c.get(url, timeout=_request_timeout(timeout), **kwargs)
|
|
102
|
+
|
|
103
|
+
service = urlsplit(public_url)
|
|
104
|
+
loopback = service.hostname in {"localhost", "127.0.0.1", "::1"}
|
|
105
|
+
if (
|
|
106
|
+
(service.scheme != "https" and not (service.scheme == "http" and loopback))
|
|
107
|
+
or not service.hostname
|
|
108
|
+
or service.username is not None
|
|
109
|
+
or service.password is not None
|
|
110
|
+
or service.query
|
|
111
|
+
or service.fragment
|
|
112
|
+
):
|
|
113
|
+
raise SourceUnavailable("BIBCITE_PUBLIC_SERVICE_URL is invalid")
|
|
114
|
+
|
|
115
|
+
params = kwargs.get("params") or {}
|
|
116
|
+
safe_params = {
|
|
117
|
+
str(key): str(value)
|
|
118
|
+
for key, value in params.items()
|
|
119
|
+
if str(key).lower() not in {"api_key", "mailto"}
|
|
120
|
+
}
|
|
121
|
+
try:
|
|
122
|
+
response = c.post(
|
|
123
|
+
public_url,
|
|
124
|
+
json={"provider": provider, "path": target.path, "params": safe_params},
|
|
125
|
+
timeout=_request_timeout(timeout),
|
|
126
|
+
follow_redirects=False,
|
|
127
|
+
)
|
|
128
|
+
except httpx.HTTPError as e:
|
|
129
|
+
raise SourceUnavailable(
|
|
130
|
+
f"public literature service unavailable ({type(e).__name__})"
|
|
131
|
+
) from e
|
|
132
|
+
if response.status_code == 404:
|
|
133
|
+
return response
|
|
134
|
+
if response.status_code == 429:
|
|
135
|
+
raise SourceUnavailable("public literature service rate-limited (429)")
|
|
136
|
+
if response.is_error:
|
|
137
|
+
raise SourceUnavailable(
|
|
138
|
+
f"public literature service error ({response.status_code})"
|
|
139
|
+
)
|
|
140
|
+
return response
|
|
84
141
|
|
|
85
142
|
|
|
86
143
|
def _sleep(delay: float):
|
|
@@ -539,62 +596,6 @@ def try_semantic_scholar(
|
|
|
539
596
|
return None
|
|
540
597
|
|
|
541
598
|
|
|
542
|
-
# ---------------------------------------------------------------------------
|
|
543
|
-
# Google Scholar (port of PaperMemory's background fetchGSData)
|
|
544
|
-
# ---------------------------------------------------------------------------
|
|
545
|
-
|
|
546
|
-
def try_google_scholar(title: str) -> Match | None:
|
|
547
|
-
with _client(browser=True) as c:
|
|
548
|
-
r = _get(
|
|
549
|
-
c,
|
|
550
|
-
"https://scholar.google.com/scholar",
|
|
551
|
-
params={"q": title, "hl": "en"},
|
|
552
|
-
)
|
|
553
|
-
if r.status_code == 429 or "captcha" in r.text.lower()[:5000]:
|
|
554
|
-
raise SourceUnavailable("Google Scholar is blocking requests (captcha/429)")
|
|
555
|
-
r.raise_for_status()
|
|
556
|
-
parts = r.text.split("gs_res_ccl_mid")
|
|
557
|
-
if len(parts) < 2:
|
|
558
|
-
return None
|
|
559
|
-
page = parts[1]
|
|
560
|
-
# Each result title anchor looks like <a id="DATAID" href=...>Title</a>
|
|
561
|
-
# (the title may contain <b> highlights and HTML entities).
|
|
562
|
-
data_id = ""
|
|
563
|
-
for am in re.finditer(
|
|
564
|
-
r'<a[^>]*\bid="([\w-]{6,40})"[^>]*>(.*?)</a>', page, re.S
|
|
565
|
-
):
|
|
566
|
-
text = html.unescape(re.sub(r"<[^>]+>", "", am.group(2)))
|
|
567
|
-
if norm_title(text) == norm_title(title):
|
|
568
|
-
data_id = am.group(1)
|
|
569
|
-
break
|
|
570
|
-
if not data_id:
|
|
571
|
-
return None
|
|
572
|
-
cite_url = (
|
|
573
|
-
"https://scholar.google.com/scholar?q=info:"
|
|
574
|
-
f"{data_id}:scholar.google.com/&output=cite&scirp=0&hl=en"
|
|
575
|
-
)
|
|
576
|
-
cite_html = _get(c, cite_url).text
|
|
577
|
-
bm = re.search(r'<a[^>]*href="([^">]+)"[^>]*>BibTex</a>', cite_html, re.I)
|
|
578
|
-
if not bm:
|
|
579
|
-
return None
|
|
580
|
-
bib_url = re.sub(r"\s+", "", bm.group(1).replace("&", "&"))
|
|
581
|
-
bibtex = _get(c, bib_url).text
|
|
582
|
-
from .bibfile import parse_bibtex_entry # local import to avoid cycle
|
|
583
|
-
|
|
584
|
-
entry = parse_bibtex_entry(bibtex)
|
|
585
|
-
venue = entry.get("journal", "") or entry.get("booktitle", "")
|
|
586
|
-
if venue and not venue.lower().endswith("xiv") and "preprint" not in venue.lower():
|
|
587
|
-
_log(f"[googlescholar] match: {venue}")
|
|
588
|
-
return Match(
|
|
589
|
-
source="googlescholar",
|
|
590
|
-
venue=venue,
|
|
591
|
-
title=clean_title(entry.get("title", title)),
|
|
592
|
-
year=entry.get("year", ""),
|
|
593
|
-
bibtex=bibtex,
|
|
594
|
-
)
|
|
595
|
-
return None
|
|
596
|
-
|
|
597
|
-
|
|
598
599
|
# ---------------------------------------------------------------------------
|
|
599
600
|
# CrossRef
|
|
600
601
|
# ---------------------------------------------------------------------------
|
|
@@ -827,7 +828,6 @@ def crossref_by_doi(doi: str) -> Match | None:
|
|
|
827
828
|
CASCADE = (
|
|
828
829
|
("dblp", lambda t, y, a, au: try_dblp(t, au)),
|
|
829
830
|
("semanticscholar", lambda t, y, a, au: try_semantic_scholar(t, y, a)),
|
|
830
|
-
("googlescholar", lambda t, y, a, au: try_google_scholar(t)),
|
|
831
831
|
("crossref", lambda t, y, a, au: try_crossref(t)),
|
|
832
832
|
("unpaywall", lambda t, y, a, au: try_unpaywall(t)),
|
|
833
833
|
("openalex", lambda t, y, a, au: try_openalex(t)),
|
|
@@ -840,8 +840,8 @@ CASCADE = (
|
|
|
840
840
|
_DISABLED: dict[str, str] = {}
|
|
841
841
|
|
|
842
842
|
# Only these sources are authoritative enough that losing one taints a miss
|
|
843
|
-
# into "incomplete".
|
|
844
|
-
#
|
|
843
|
+
# into "incomplete". Unpaywall flakiness is routine and must not stop
|
|
844
|
+
# "not_found" from ever being trustworthy.
|
|
845
845
|
# Override with BIBCITE_CORE_SOURCES="dblp,semanticscholar" if one of these
|
|
846
846
|
# is down for days and keeps every verdict incomplete.
|
|
847
847
|
CORE_SOURCES = frozenset(
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Parses the vendored ``data/strings.bib`` @string table (journals /
|
|
4
4
|
conferences / workshops) and maps venue strings returned by DBLP, Semantic
|
|
5
|
-
Scholar,
|
|
5
|
+
Scholar, CrossRef, Unpaywall, etc. onto the canonical names.
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
8
|
import re
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Exercise the CLI publication cascade without contacting external services."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import json
|
|
5
|
+
|
|
6
|
+
import httpx
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from bibcite import cache, cli, sources
|
|
10
|
+
|
|
11
|
+
resolver = importlib.import_module("bibcite.resolve")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@pytest.mark.parametrize("operation", ["get", "add", "upgrade"])
|
|
15
|
+
def test_cli_never_queries_google_scholar(operation, monkeypatch, tmp_path, capsys):
|
|
16
|
+
monkeypatch.setattr(cache, "DISABLED", True)
|
|
17
|
+
monkeypatch.setattr(sources, "_DISABLED", {})
|
|
18
|
+
monkeypatch.setattr(
|
|
19
|
+
resolver,
|
|
20
|
+
"arxiv_metadata",
|
|
21
|
+
lambda _: sources.ArxivMeta(
|
|
22
|
+
"1706.03762",
|
|
23
|
+
"Attention Is All You Need",
|
|
24
|
+
["Ashish Vaswani"],
|
|
25
|
+
"2017",
|
|
26
|
+
"https://arxiv.org/abs/1706.03762",
|
|
27
|
+
),
|
|
28
|
+
)
|
|
29
|
+
visited = []
|
|
30
|
+
for name in ("dblp", "semantic_scholar", "crossref", "unpaywall", "openalex"):
|
|
31
|
+
|
|
32
|
+
def miss(*args, source=name):
|
|
33
|
+
visited.append(source)
|
|
34
|
+
return None
|
|
35
|
+
|
|
36
|
+
monkeypatch.setattr(sources, f"try_{name}", miss)
|
|
37
|
+
monkeypatch.setattr(sources, "try_dblp_fuzzy", lambda *args: None)
|
|
38
|
+
requests = []
|
|
39
|
+
|
|
40
|
+
def unexpected_request(client, url, **kwargs):
|
|
41
|
+
requests.append(url)
|
|
42
|
+
return httpx.Response(429, request=httpx.Request("GET", url))
|
|
43
|
+
|
|
44
|
+
monkeypatch.setattr(sources, "_get", unexpected_request)
|
|
45
|
+
path = tmp_path / "references.bib"
|
|
46
|
+
if operation == "get":
|
|
47
|
+
args = ["get", "--json", "1706.03762"]
|
|
48
|
+
elif operation == "add":
|
|
49
|
+
args = ["add", "--no-tidy", str(path), "1706.03762"]
|
|
50
|
+
else:
|
|
51
|
+
path.write_text("""@article{vaswani2017attention,
|
|
52
|
+
title = {Attention Is All You Need},
|
|
53
|
+
author = {Ashish Vaswani},
|
|
54
|
+
year = {2017},
|
|
55
|
+
journal = {arXiv preprint arXiv:1706.03762},
|
|
56
|
+
eprint = {1706.03762}
|
|
57
|
+
}
|
|
58
|
+
""")
|
|
59
|
+
args = ["upgrade", str(path), "--dry-run"]
|
|
60
|
+
cli.main(args)
|
|
61
|
+
json.loads(capsys.readouterr().out)
|
|
62
|
+
assert set(visited) == {
|
|
63
|
+
"dblp",
|
|
64
|
+
"semantic_scholar",
|
|
65
|
+
"crossref",
|
|
66
|
+
"unpaywall",
|
|
67
|
+
"openalex",
|
|
68
|
+
}
|
|
69
|
+
assert requests == []
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
import httpx
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from bibcite import sources
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@pytest.fixture(autouse=True)
|
|
10
|
+
def clean_environment(monkeypatch):
|
|
11
|
+
for name in (
|
|
12
|
+
"BIBCITE_PUBLIC_SERVICE_URL",
|
|
13
|
+
"BIBCITE_MAILTO",
|
|
14
|
+
"OPENALEX_API_KEY",
|
|
15
|
+
"S2_API_KEY",
|
|
16
|
+
"SEMANTIC_SCHOLAR_API_KEY",
|
|
17
|
+
):
|
|
18
|
+
monkeypatch.delenv(name, raising=False)
|
|
19
|
+
monkeypatch.setattr(sources, "_LAST_REQUEST", {})
|
|
20
|
+
monkeypatch.setattr(sources.time, "sleep", lambda _: None)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _client(handler):
|
|
24
|
+
return httpx.Client(transport=httpx.MockTransport(handler))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@pytest.mark.parametrize(
|
|
28
|
+
("url", "provider", "path"),
|
|
29
|
+
(
|
|
30
|
+
("https://api.openalex.org/works", "openalex", "/works"),
|
|
31
|
+
(
|
|
32
|
+
"https://api.semanticscholar.org/graph/v1/paper/search",
|
|
33
|
+
"semanticscholar",
|
|
34
|
+
"/graph/v1/paper/search",
|
|
35
|
+
),
|
|
36
|
+
(
|
|
37
|
+
"https://api.crossref.org/works/10.1234/example/transform/application/x-bibtex",
|
|
38
|
+
"crossref",
|
|
39
|
+
"/works/10.1234/example/transform/application/x-bibtex",
|
|
40
|
+
),
|
|
41
|
+
),
|
|
42
|
+
)
|
|
43
|
+
def test_keyless_supported_requests_route_to_public_service(
|
|
44
|
+
url, provider, path, monkeypatch
|
|
45
|
+
):
|
|
46
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
47
|
+
requests = []
|
|
48
|
+
|
|
49
|
+
def handler(request):
|
|
50
|
+
requests.append(request)
|
|
51
|
+
return httpx.Response(200, json={"unchanged": True})
|
|
52
|
+
|
|
53
|
+
with _client(handler) as client:
|
|
54
|
+
response = sources._get(
|
|
55
|
+
client,
|
|
56
|
+
url,
|
|
57
|
+
params={
|
|
58
|
+
"query": "paper",
|
|
59
|
+
"limit": 5,
|
|
60
|
+
"api_key": "secret",
|
|
61
|
+
"mailto": "secret@example.com",
|
|
62
|
+
},
|
|
63
|
+
headers={"x-api-key": "secret"},
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
assert response.json() == {"unchanged": True}
|
|
67
|
+
assert len(requests) == 1
|
|
68
|
+
request = requests[0]
|
|
69
|
+
assert request.method == "POST"
|
|
70
|
+
assert request.url == "https://literature.test/v1/query"
|
|
71
|
+
assert "x-api-key" not in request.headers
|
|
72
|
+
payload = json.loads(request.content)
|
|
73
|
+
assert payload == {
|
|
74
|
+
"provider": provider,
|
|
75
|
+
"path": path,
|
|
76
|
+
"params": {"query": "paper", "limit": "5"},
|
|
77
|
+
}
|
|
78
|
+
assert "secret" not in request.content.decode()
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@pytest.mark.parametrize(
|
|
82
|
+
("variable", "url", "params"),
|
|
83
|
+
(
|
|
84
|
+
("OPENALEX_API_KEY", "https://api.openalex.org/works", {"api_key": "mine"}),
|
|
85
|
+
("S2_API_KEY", "https://api.semanticscholar.org/graph/v1/paper/search", {}),
|
|
86
|
+
(
|
|
87
|
+
"BIBCITE_MAILTO",
|
|
88
|
+
"https://api.crossref.org/works",
|
|
89
|
+
{"mailto": "me@example.com"},
|
|
90
|
+
),
|
|
91
|
+
),
|
|
92
|
+
)
|
|
93
|
+
def test_personal_access_overrides_public_service(variable, url, params, monkeypatch):
|
|
94
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
95
|
+
monkeypatch.setenv(variable, "mine")
|
|
96
|
+
requests = []
|
|
97
|
+
|
|
98
|
+
def handler(request):
|
|
99
|
+
requests.append(request)
|
|
100
|
+
return httpx.Response(200, json={})
|
|
101
|
+
|
|
102
|
+
with _client(handler) as client:
|
|
103
|
+
sources._get(client, url, params=params)
|
|
104
|
+
|
|
105
|
+
assert len(requests) == 1
|
|
106
|
+
assert requests[0].method == "GET"
|
|
107
|
+
assert requests[0].url.host != "literature.test"
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def test_without_public_service_requests_remain_direct():
|
|
111
|
+
requests = []
|
|
112
|
+
|
|
113
|
+
def handler(request):
|
|
114
|
+
requests.append(request)
|
|
115
|
+
return httpx.Response(200, json={})
|
|
116
|
+
|
|
117
|
+
with _client(handler) as client:
|
|
118
|
+
sources._get(client, "https://api.openalex.org/works", params={"search": "x"})
|
|
119
|
+
|
|
120
|
+
assert len(requests) == 1
|
|
121
|
+
assert requests[0].method == "GET"
|
|
122
|
+
assert requests[0].url.host == "api.openalex.org"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@pytest.mark.parametrize("status", [429, 500])
|
|
126
|
+
def test_public_service_http_failure_is_not_retried(status, monkeypatch):
|
|
127
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
128
|
+
calls = 0
|
|
129
|
+
|
|
130
|
+
def handler(request):
|
|
131
|
+
nonlocal calls
|
|
132
|
+
calls += 1
|
|
133
|
+
return httpx.Response(status)
|
|
134
|
+
|
|
135
|
+
with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
|
|
136
|
+
sources._paced_get(
|
|
137
|
+
client,
|
|
138
|
+
"https://api.semanticscholar.org/graph/v1/paper/search",
|
|
139
|
+
"semanticscholar",
|
|
140
|
+
0,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
assert calls == 1
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def test_public_service_timeout_is_not_retried(monkeypatch):
|
|
147
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
148
|
+
calls = 0
|
|
149
|
+
|
|
150
|
+
def handler(request):
|
|
151
|
+
nonlocal calls
|
|
152
|
+
calls += 1
|
|
153
|
+
raise httpx.ReadTimeout("timed out", request=request)
|
|
154
|
+
|
|
155
|
+
with _client(handler) as client, pytest.raises(sources.SourceUnavailable):
|
|
156
|
+
sources._paced_get(
|
|
157
|
+
client,
|
|
158
|
+
"https://api.semanticscholar.org/graph/v1/paper/search",
|
|
159
|
+
"semanticscholar",
|
|
160
|
+
0,
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
assert calls == 1
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_public_service_404_is_preserved(monkeypatch):
|
|
167
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", "https://literature.test/v1/query")
|
|
168
|
+
|
|
169
|
+
with _client(lambda request: httpx.Response(404)) as client:
|
|
170
|
+
response = sources._get(client, "https://api.openalex.org/works/W1")
|
|
171
|
+
|
|
172
|
+
assert response.status_code == 404
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@pytest.mark.parametrize(
|
|
176
|
+
"value",
|
|
177
|
+
(
|
|
178
|
+
"http://literature.test/v1/query",
|
|
179
|
+
"https://user:pass@literature.test/v1/query",
|
|
180
|
+
"https://literature.test/v1/query?secret=x",
|
|
181
|
+
"https://literature.test/v1/query#fragment",
|
|
182
|
+
),
|
|
183
|
+
)
|
|
184
|
+
def test_invalid_public_service_url_is_rejected(value, monkeypatch):
|
|
185
|
+
monkeypatch.setenv("BIBCITE_PUBLIC_SERVICE_URL", value)
|
|
186
|
+
with _client(lambda request: httpx.Response(200)) as client:
|
|
187
|
+
with pytest.raises(sources.SourceUnavailable):
|
|
188
|
+
sources._get(client, "https://api.openalex.org/works")
|
|
@@ -43,7 +43,7 @@ def test_core_source_429_taints_verdict(monkeypatch):
|
|
|
43
43
|
monkeypatch.setattr(
|
|
44
44
|
sources,
|
|
45
45
|
"CASCADE",
|
|
46
|
-
_cascade(dblp="raise",
|
|
46
|
+
_cascade(dblp="raise", unpaywall=None, crossref=None),
|
|
47
47
|
)
|
|
48
48
|
match, status = find_published("Some Title", author_hint="smith")
|
|
49
49
|
assert (match, status) == (None, "incomplete")
|
|
@@ -60,11 +60,11 @@ def test_previously_disabled_core_source_taints_next_queries(monkeypatch):
|
|
|
60
60
|
|
|
61
61
|
|
|
62
62
|
def test_noncore_outage_does_not_taint(monkeypatch):
|
|
63
|
-
#
|
|
63
|
+
# Unpaywall outages are routine; a miss stays trustworthy.
|
|
64
64
|
monkeypatch.setattr(
|
|
65
65
|
sources,
|
|
66
66
|
"CASCADE",
|
|
67
|
-
_cascade(dblp=None,
|
|
67
|
+
_cascade(dblp=None, unpaywall="raise", crossref=None),
|
|
68
68
|
)
|
|
69
69
|
match, status = find_published("Some Title", author_hint="smith")
|
|
70
70
|
assert (match, status) == (None, "not_found")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|