bibcite-cli 0.6.0__tar.gz → 0.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/PKG-INFO +8 -3
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/Readme.md +7 -2
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/pyproject.toml +1 -1
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/resolve.py +45 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/sources.py +202 -36
- bibcite_cli-0.6.2/tests/test_webpages.py +69 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/uv.lock +2 -2
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/.github/workflows/ci.yml +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/.github/workflows/publish.yml +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/.gitignore +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/LICENSE +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/assets/bibcite.svg +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/skills/bibcite/SKILL.md +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/__init__.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/bibfile.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/cli.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/normalize.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/venues.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_cli_status.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_round2.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_round3.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_source_retries.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_status_semantics.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_venues.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.2
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -18,7 +18,7 @@ Description-Content-Type: text/markdown
|
|
|
18
18
|
<h1 align="center">bibcite</h1>
|
|
19
19
|
|
|
20
20
|
<p align="center">
|
|
21
|
-
Turn an arXiv ID, DOI,
|
|
21
|
+
Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
|
|
22
22
|
</p>
|
|
23
23
|
|
|
24
24
|
<p align="center">
|
|
@@ -105,7 +105,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
|
|
|
105
105
|
|
|
106
106
|
## What it handles
|
|
107
107
|
|
|
108
|
-
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and
|
|
108
|
+
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
|
|
109
|
+
- A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
|
|
109
110
|
- It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
|
|
110
111
|
- It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
|
|
111
112
|
- It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
|
|
@@ -208,6 +209,10 @@ Install the checkout as an editable command while developing:
|
|
|
208
209
|
uv tool install --editable .
|
|
209
210
|
```
|
|
210
211
|
|
|
212
|
+
## Acknowledgements
|
|
213
|
+
|
|
214
|
+
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
215
|
+
|
|
211
216
|
## License
|
|
212
217
|
|
|
213
218
|
`bibcite` is available under the [MIT License](LICENSE).
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
<h1 align="center">bibcite</h1>
|
|
6
6
|
|
|
7
7
|
<p align="center">
|
|
8
|
-
Turn an arXiv ID, DOI,
|
|
8
|
+
Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
|
|
9
9
|
</p>
|
|
10
10
|
|
|
11
11
|
<p align="center">
|
|
@@ -92,7 +92,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
|
|
|
92
92
|
|
|
93
93
|
## What it handles
|
|
94
94
|
|
|
95
|
-
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and
|
|
95
|
+
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
|
|
96
|
+
- A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
|
|
96
97
|
- It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
|
|
97
98
|
- It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
|
|
98
99
|
- It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
|
|
@@ -195,6 +196,10 @@ Install the checkout as an editable command while developing:
|
|
|
195
196
|
uv tool install --editable .
|
|
196
197
|
```
|
|
197
198
|
|
|
199
|
+
## Acknowledgements
|
|
200
|
+
|
|
201
|
+
Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
|
|
202
|
+
|
|
198
203
|
## License
|
|
199
204
|
|
|
200
205
|
`bibcite` is available under the [MIT License](LICENSE).
|
|
@@ -22,6 +22,9 @@ from .sources import (
|
|
|
22
22
|
Match,
|
|
23
23
|
arxiv_metadata,
|
|
24
24
|
crossref_by_doi,
|
|
25
|
+
PageUnreadable,
|
|
26
|
+
WebPage,
|
|
27
|
+
fetch_web_page,
|
|
25
28
|
find_published,
|
|
26
29
|
)
|
|
27
30
|
from .venues import canonicalize
|
|
@@ -43,6 +46,7 @@ ARXIV_URL = re.compile(
|
|
|
43
46
|
re.I,
|
|
44
47
|
)
|
|
45
48
|
DOI_URL = re.compile(r"doi\.org/(10\.\S+)", re.I)
|
|
49
|
+
WEB_URL = re.compile(r"^https?://\S+$", re.I)
|
|
46
50
|
DOI_RE = re.compile(r"^10\.\d{4,9}/\S+$")
|
|
47
51
|
ARXIV_DOI = re.compile(
|
|
48
52
|
r"^10\.48550/arxiv\."
|
|
@@ -74,6 +78,12 @@ def classify(query: str) -> tuple[str, str]:
|
|
|
74
78
|
return "arxiv", m.group(1)
|
|
75
79
|
if DOI_RE.match(q):
|
|
76
80
|
return "doi", q
|
|
81
|
+
# A URL that is not arXiv and not a DOI is a page: a blog post, a standard,
|
|
82
|
+
# a documentation page. Nothing indexes those, so searching for it as a
|
|
83
|
+
# title only ever returns "no match anywhere" — the page itself is the
|
|
84
|
+
# source.
|
|
85
|
+
if WEB_URL.match(q):
|
|
86
|
+
return "webpage", q
|
|
77
87
|
return "title", query.strip()
|
|
78
88
|
|
|
79
89
|
|
|
@@ -198,6 +208,31 @@ def _arxiv_only_entry(meta: ArxivMeta) -> dict:
|
|
|
198
208
|
}
|
|
199
209
|
|
|
200
210
|
|
|
211
|
+
def _web_entry(page: WebPage) -> dict:
|
|
212
|
+
"""A page cited as @misc, the one type every conference .bst understands.
|
|
213
|
+
|
|
214
|
+
@online is biblatex-only; a NeurIPS or IEEE style would drop the entry
|
|
215
|
+
entirely. howpublished carries the link for the same reason the arXiv
|
|
216
|
+
preprint entry uses it: classic styles print it and ignore `url`. No access
|
|
217
|
+
date, because the tidy step omits `note` from every entry in the file.
|
|
218
|
+
"""
|
|
219
|
+
entry = {
|
|
220
|
+
"ENTRYTYPE": "misc",
|
|
221
|
+
"title": page.title,
|
|
222
|
+
"howpublished": f"\\url{{{page.url}}}",
|
|
223
|
+
"url": page.url,
|
|
224
|
+
}
|
|
225
|
+
if page.authors:
|
|
226
|
+
entry["author"] = " and ".join(page.authors)
|
|
227
|
+
elif page.site:
|
|
228
|
+
# No byline: the site is the closest thing to a corporate author, and a
|
|
229
|
+
# key of `anonymousXXXX…` is worse than one naming where it came from.
|
|
230
|
+
entry["author"] = f"{{{page.site}}}"
|
|
231
|
+
if page.year:
|
|
232
|
+
entry["year"] = page.year
|
|
233
|
+
return entry
|
|
234
|
+
|
|
235
|
+
|
|
201
236
|
def resolve(query: str, require_published: bool = False) -> Resolved:
|
|
202
237
|
kind, value = classify(query)
|
|
203
238
|
_log(f"[bibcite] query understood as {kind}: {value}")
|
|
@@ -248,6 +283,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
|
|
|
248
283
|
entry = _arxiv_only_entry(meta)
|
|
249
284
|
return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
|
|
250
285
|
|
|
286
|
+
if kind == "webpage":
|
|
287
|
+
try:
|
|
288
|
+
page = fetch_web_page(value)
|
|
289
|
+
except PageUnreadable as e:
|
|
290
|
+
raise NotFound(f"could not read {value}: {e}") from e
|
|
291
|
+
if not page.title:
|
|
292
|
+
raise NotFound(f"No title found on the page: {value}")
|
|
293
|
+
_log(f"[web] {page.title}" + (f" ({page.year})" if page.year else ""))
|
|
294
|
+
return Resolved(_finalize(_web_entry(page), None), "webpage", page.site, False)
|
|
295
|
+
|
|
251
296
|
if kind == "doi":
|
|
252
297
|
match = crossref_by_doi(value)
|
|
253
298
|
if not match or not match.title:
|
|
@@ -12,6 +12,7 @@ import re
|
|
|
12
12
|
import sys
|
|
13
13
|
import time
|
|
14
14
|
import xml.etree.ElementTree as ET
|
|
15
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
15
16
|
from dataclasses import dataclass, field
|
|
16
17
|
|
|
17
18
|
import httpx
|
|
@@ -23,7 +24,12 @@ BROWSER_UA = (
|
|
|
23
24
|
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
|
|
24
25
|
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
|
25
26
|
)
|
|
26
|
-
|
|
27
|
+
# Per-request cap for the publication-matching sources. Kept short so one slow
|
|
28
|
+
# or half-down source (DBLP in particular) can't stall an interactive resolve;
|
|
29
|
+
# a source that can't answer in time is treated as unavailable → "incomplete",
|
|
30
|
+
# never a false "not published". The arXiv metadata fetch sets its own longer
|
|
31
|
+
# timeout on the request itself, so this does not affect it.
|
|
32
|
+
TIMEOUT = 8.0
|
|
27
33
|
|
|
28
34
|
PREPRINT_VENUES = re.compile(r"arxiv|corr|biorxiv|medrxiv|chemrxiv|ssrn|preprint", re.I)
|
|
29
35
|
ARXIV_DOI = re.compile(r"^10\.48550/", re.I)
|
|
@@ -37,6 +43,12 @@ class SourceUnavailable(Exception):
|
|
|
37
43
|
"""Raised when a source rate-limits/blocks us; the cascade skips it."""
|
|
38
44
|
|
|
39
45
|
|
|
46
|
+
class PageUnreadable(Exception):
|
|
47
|
+
"""A page refused to be read and will refuse again — a different
|
|
48
|
+
identifier, or a hand-written entry, is the next step rather than a
|
|
49
|
+
retry."""
|
|
50
|
+
|
|
51
|
+
|
|
40
52
|
class TransientSourceError(SourceUnavailable):
|
|
41
53
|
"""Raised after request retries are exhausted without tripping the
|
|
42
54
|
process-wide circuit breaker for later batch entries."""
|
|
@@ -183,30 +195,32 @@ def _paced_get(
|
|
|
183
195
|
params: dict | None = None,
|
|
184
196
|
headers: dict | None = None,
|
|
185
197
|
) -> httpx.Response:
|
|
186
|
-
for attempt in range(
|
|
198
|
+
for attempt in range(2):
|
|
187
199
|
wait = min_interval - (time.monotonic() - _LAST_REQUEST.get(source, 0.0))
|
|
188
200
|
if wait > 0:
|
|
189
201
|
time.sleep(wait)
|
|
190
202
|
_LAST_REQUEST[source] = time.monotonic()
|
|
191
203
|
try:
|
|
192
204
|
r = c.get(url, params=params, headers=headers)
|
|
193
|
-
except httpx.HTTPError as e: # Retry transport errors before failing
|
|
194
|
-
if attempt <
|
|
195
|
-
time.sleep(
|
|
205
|
+
except httpx.HTTPError as e: # Retry transport errors once before failing.
|
|
206
|
+
if attempt < 1:
|
|
207
|
+
time.sleep(1)
|
|
196
208
|
continue
|
|
197
209
|
raise TransientSourceError(
|
|
198
210
|
f"{source} unreachable ({type(e).__name__})"
|
|
199
211
|
) from e
|
|
200
212
|
if r.status_code == 429:
|
|
213
|
+
# A 429 is usually a persistent rate-limit (e.g. the shared
|
|
214
|
+
# unauthenticated Semantic Scholar pool), not a transient blip, so a
|
|
215
|
+
# long client-side backoff rarely clears it and just stalls an
|
|
216
|
+
# interactive resolve. Take at most one quick retry when the server
|
|
217
|
+
# asks for a short wait, then give up and let the circuit breaker
|
|
218
|
+
# skip this source for the rest of the run.
|
|
201
219
|
retry_after = int(r.headers.get("Retry-After") or 0)
|
|
202
|
-
if retry_after
|
|
203
|
-
|
|
204
|
-
if attempt < 2:
|
|
205
|
-
delay = max(retry_after, 4 * (attempt + 1))
|
|
206
|
-
_log(f"[{source}] 429 — backing off {delay}s")
|
|
207
|
-
time.sleep(delay)
|
|
220
|
+
if attempt < 1 and retry_after <= 2:
|
|
221
|
+
time.sleep(max(retry_after, 1))
|
|
208
222
|
continue
|
|
209
|
-
raise SourceUnavailable(f"{source} rate-limited (429)
|
|
223
|
+
raise SourceUnavailable(f"{source} rate-limited (429)")
|
|
210
224
|
return r
|
|
211
225
|
raise SourceUnavailable(f"{source} unavailable")
|
|
212
226
|
|
|
@@ -795,32 +809,51 @@ def find_published(
|
|
|
795
809
|
_log(f"[cache] hit: {cached.get('venue', '')} ({cached.get('source', '')})")
|
|
796
810
|
return Match(**cached), "found"
|
|
797
811
|
|
|
798
|
-
clean_misses = 0
|
|
799
812
|
# Core sources lost earlier in this run taint this query's verdict too.
|
|
800
813
|
incomplete = any(n in CORE_SOURCES for n in _DISABLED)
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
814
|
+
# Query every still-viable source concurrently. A preprint with no published
|
|
815
|
+
# version (the common case) misses everywhere, and used to pay the *sum* of
|
|
816
|
+
# each source's latency; now the wall-clock is the slowest single source.
|
|
817
|
+
# The first verified hit by CASCADE priority still wins.
|
|
818
|
+
active = [(name, fn) for name, fn in CASCADE if name not in _DISABLED]
|
|
819
|
+
outcomes: dict[str, tuple] = {}
|
|
820
|
+
if active:
|
|
821
|
+
with ThreadPoolExecutor(max_workers=len(active)) as pool:
|
|
822
|
+
futures = {
|
|
823
|
+
pool.submit(fn, title, year, arxiv_id, author_hint): name
|
|
824
|
+
for name, fn in active
|
|
825
|
+
}
|
|
826
|
+
for future in as_completed(futures):
|
|
827
|
+
name = futures[future]
|
|
828
|
+
try:
|
|
829
|
+
m = future.result()
|
|
830
|
+
if m:
|
|
831
|
+
outcomes[name] = ("found", m)
|
|
832
|
+
else:
|
|
833
|
+
outcomes[name] = ("miss", None)
|
|
834
|
+
_log(f"[{name}] no publication found")
|
|
835
|
+
except TransientSourceError as e:
|
|
836
|
+
outcomes[name] = ("fail", name in CORE_SOURCES)
|
|
837
|
+
_log(f"[{name}] transient failure for this entry: {e}")
|
|
838
|
+
except SourceUnavailable as e:
|
|
839
|
+
_DISABLED[name] = str(e)
|
|
840
|
+
outcomes[name] = ("fail", name in CORE_SOURCES)
|
|
841
|
+
_log(f"[{name}] disabled for the rest of this run: {e}")
|
|
842
|
+
except Exception as e: # a hiccup on one source must not kill the run
|
|
843
|
+
outcomes[name] = ("fail", name in CORE_SOURCES)
|
|
844
|
+
_log(f"[{name}] error: {type(e).__name__}: {e}")
|
|
845
|
+
|
|
846
|
+
# Prefer the highest-priority source that verified a match.
|
|
847
|
+
for name, _ in CASCADE:
|
|
848
|
+
outcome = outcomes.get(name)
|
|
849
|
+
if outcome and outcome[0] == "found":
|
|
850
|
+
cache.put(cache_key, outcome[1].__dict__)
|
|
851
|
+
return outcome[1], "found"
|
|
852
|
+
|
|
853
|
+
clean_misses = sum(1 for outcome in outcomes.values() if outcome[0] == "miss")
|
|
854
|
+
incomplete = incomplete or any(
|
|
855
|
+
outcome[0] == "fail" and outcome[1] for outcome in outcomes.values()
|
|
856
|
+
)
|
|
824
857
|
|
|
825
858
|
# Exact-title search missed everywhere. Before concluding "no published
|
|
826
859
|
# version", try the title-drift fallback — camera-ready titles frequently
|
|
@@ -844,3 +877,136 @@ def find_published(
|
|
|
844
877
|
if not clean_misses:
|
|
845
878
|
return None, "unavailable"
|
|
846
879
|
return None, ("incomplete" if incomplete else "not_found")
|
|
880
|
+
|
|
881
|
+
|
|
882
|
+
@dataclass
|
|
883
|
+
class WebPage:
|
|
884
|
+
"""What a web page can say about itself, for citing it as an @misc."""
|
|
885
|
+
|
|
886
|
+
url: str
|
|
887
|
+
title: str
|
|
888
|
+
authors: list[str] = field(default_factory=list)
|
|
889
|
+
year: str = ""
|
|
890
|
+
site: str = ""
|
|
891
|
+
|
|
892
|
+
|
|
893
|
+
def _meta_content(html_text: str, *keys: str) -> str:
|
|
894
|
+
"""The content of the first <meta> whose name/property matches a key.
|
|
895
|
+
|
|
896
|
+
Attribute order varies between generators, so both orders are tried rather
|
|
897
|
+
than assuming content comes last.
|
|
898
|
+
"""
|
|
899
|
+
for key in keys:
|
|
900
|
+
for pattern in (
|
|
901
|
+
rf'<meta[^>]+(?:name|property)=["\']{re.escape(key)}["\'][^>]*'
|
|
902
|
+
rf'content=["\'](.*?)["\']',
|
|
903
|
+
rf'<meta[^>]+content=["\'](.*?)["\'][^>]*'
|
|
904
|
+
rf'(?:name|property)=["\']{re.escape(key)}["\']',
|
|
905
|
+
):
|
|
906
|
+
m = re.search(pattern, html_text, re.I | re.S)
|
|
907
|
+
if m and m.group(1).strip():
|
|
908
|
+
return html.unescape(m.group(1).strip())
|
|
909
|
+
return ""
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
def fetch_web_page(url: str) -> WebPage:
|
|
913
|
+
"""Read a page's own description of itself.
|
|
914
|
+
|
|
915
|
+
Blogs, documentation and standards pages are cited constantly and are in
|
|
916
|
+
none of the academic indexes, so there is nothing to look them up in — the
|
|
917
|
+
page itself is the only source. Highwire and Dublin Core tags come first
|
|
918
|
+
because sites that carry them mean them; Open Graph and <title> are the
|
|
919
|
+
fallback every site has.
|
|
920
|
+
"""
|
|
921
|
+
try:
|
|
922
|
+
with _client(browser=True) as client:
|
|
923
|
+
response = client.get(url)
|
|
924
|
+
response.raise_for_status()
|
|
925
|
+
body = response.text[:400_000]
|
|
926
|
+
except httpx.HTTPStatusError as e:
|
|
927
|
+
status = e.response.status_code
|
|
928
|
+
# 403 and 404 are settled answers: the page will not become readable on
|
|
929
|
+
# a retry, so say so rather than sending the caller back to wait.
|
|
930
|
+
if status in (401, 403, 404, 410) or 400 <= status < 500:
|
|
931
|
+
raise PageUnreadable(
|
|
932
|
+
f"the page answered {status} — cite it with --bibtex, or use its DOI if it has one"
|
|
933
|
+
) from e
|
|
934
|
+
raise SourceUnavailable(f"could not fetch {url}: {e}") from e
|
|
935
|
+
except httpx.HTTPError as e:
|
|
936
|
+
raise SourceUnavailable(f"could not fetch {url}: {e}") from e
|
|
937
|
+
|
|
938
|
+
title = (
|
|
939
|
+
_meta_content(body, "citation_title", "DC.title", "og:title", "twitter:title")
|
|
940
|
+
or _title_tag(body)
|
|
941
|
+
)
|
|
942
|
+
authors = [
|
|
943
|
+
author
|
|
944
|
+
for author in (
|
|
945
|
+
_meta_content(body, "citation_author", "DC.creator", "author", "article:author"),
|
|
946
|
+
)
|
|
947
|
+
if author
|
|
948
|
+
]
|
|
949
|
+
date = _meta_content(
|
|
950
|
+
body,
|
|
951
|
+
"citation_publication_date",
|
|
952
|
+
"citation_date",
|
|
953
|
+
"DC.date",
|
|
954
|
+
"article:published_time",
|
|
955
|
+
"og:updated_time",
|
|
956
|
+
"date",
|
|
957
|
+
)
|
|
958
|
+
year = _year_in(date) or _year_in_path(url)
|
|
959
|
+
site = _meta_content(body, "og:site_name") or _site_author(_host(url))
|
|
960
|
+
return WebPage(url=str(response.url), title=title, authors=authors, year=year, site=site)
|
|
961
|
+
|
|
962
|
+
|
|
963
|
+
def _year_in(text: str) -> str:
|
|
964
|
+
m = re.search(r"(?:19|20)\d{2}", text)
|
|
965
|
+
return m.group(0) if m else ""
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
def _year_in_path(url: str) -> str:
|
|
969
|
+
"""The year a dateless page puts in its own URL.
|
|
970
|
+
|
|
971
|
+
Blog engines write `/2015/05/21/title`, and a citation key of
|
|
972
|
+
`karpathyXXXXunreasonable` is worse than one carrying the year the post
|
|
973
|
+
announces about itself. Only the path is read: a query string can hold any
|
|
974
|
+
number at all.
|
|
975
|
+
"""
|
|
976
|
+
path = re.sub(r"^https?://[^/]+", "", url).split("?")[0].split("#")[0]
|
|
977
|
+
for segment in path.split("/"):
|
|
978
|
+
if re.fullmatch(r"(?:19|20)\d{2}", segment):
|
|
979
|
+
return segment
|
|
980
|
+
return ""
|
|
981
|
+
|
|
982
|
+
|
|
983
|
+
def _title_tag(html_text: str) -> str:
|
|
984
|
+
m = re.search(r"<title[^>]*>(.*?)</title>", html_text, re.I | re.S)
|
|
985
|
+
if not m:
|
|
986
|
+
return ""
|
|
987
|
+
# Strip the trailing " — Site Name" many templates append; the site name is
|
|
988
|
+
# recorded separately, and repeating it in the title reads badly in a
|
|
989
|
+
# bibliography.
|
|
990
|
+
title = html.unescape(re.sub(r"\s+", " ", m.group(1))).strip()
|
|
991
|
+
return re.sub(r"\s*[|·—–-]\s*[^|·—–-]{1,40}$", "", title).strip() or title
|
|
992
|
+
|
|
993
|
+
|
|
994
|
+
def _host(url: str) -> str:
|
|
995
|
+
m = re.match(r"https?://(?:www\.)?([^/:]+)", url, re.I)
|
|
996
|
+
return m.group(1) if m else ""
|
|
997
|
+
|
|
998
|
+
|
|
999
|
+
def _site_author(host: str) -> str:
|
|
1000
|
+
"""A readable stand-in author for a page with no byline.
|
|
1001
|
+
|
|
1002
|
+
The bare host makes an unreadable key — `karpathygithubioXXXXunreasonable`
|
|
1003
|
+
— so the hosting suffix goes and the name that identifies the site stays:
|
|
1004
|
+
`karpathy.github.io` reads as Karpathy, `docs.python.org` as Python.
|
|
1005
|
+
"""
|
|
1006
|
+
labels = [label for label in host.lower().split(".") if label]
|
|
1007
|
+
if not labels:
|
|
1008
|
+
return host
|
|
1009
|
+
generic = {"github", "io", "com", "org", "net", "edu", "gov", "ai", "dev",
|
|
1010
|
+
"co", "uk", "cn", "blog", "www", "docs", "pages", "medium"}
|
|
1011
|
+
named = [label for label in labels if label not in generic]
|
|
1012
|
+
return (named[0] if named else labels[0]).capitalize()
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Citing a page that no index knows about.
|
|
2
|
+
|
|
3
|
+
Blogs, documentation and standards pages are cited constantly and appear in
|
|
4
|
+
none of the academic sources, so the page itself has to be the source.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from bibcite.resolve import _web_entry, classify
|
|
8
|
+
from bibcite.sources import WebPage, _meta_content, _site_author, _title_tag, _year_in_path
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_a_plain_url_is_a_webpage_and_the_others_still_are_not():
|
|
12
|
+
assert classify("https://karpathy.github.io/2015/05/21/rnn-effectiveness/")[0] == "webpage"
|
|
13
|
+
assert classify("http://example.edu/notes")[0] == "webpage"
|
|
14
|
+
# The identifiers that resolve properly must keep their own paths.
|
|
15
|
+
assert classify("https://arxiv.org/abs/1706.03762") == ("arxiv", "1706.03762")
|
|
16
|
+
assert classify("https://doi.org/10.1038/s41586-021-03819-2")[0] == "doi"
|
|
17
|
+
assert classify("10.1109/CVPR.2016.90")[0] == "doi"
|
|
18
|
+
assert classify("Attention Is All You Need")[0] == "title"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_entry_is_misc_so_conference_styles_can_print_it():
|
|
22
|
+
# @online is biblatex-only: a NeurIPS or IEEE .bst drops the entry.
|
|
23
|
+
entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", year="2024"))
|
|
24
|
+
assert entry["ENTRYTYPE"] == "misc"
|
|
25
|
+
assert entry["howpublished"] == r"\url{https://example.org/post}"
|
|
26
|
+
assert entry["url"] == "https://example.org/post"
|
|
27
|
+
assert entry["year"] == "2024"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_a_page_with_no_byline_is_attributed_to_its_site():
|
|
31
|
+
entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", site="Example"))
|
|
32
|
+
# Braced, so the .bst treats it as a corporate name and does not invert it.
|
|
33
|
+
assert entry["author"] == "{Example}"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_a_byline_wins_over_the_site():
|
|
37
|
+
entry = _web_entry(
|
|
38
|
+
WebPage(url="https://example.org/post", title="A Post", authors=["Ada Lovelace"], site="Example")
|
|
39
|
+
)
|
|
40
|
+
assert entry["author"] == "Ada Lovelace"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def test_the_site_author_drops_hosting_suffixes():
|
|
44
|
+
# `karpathygithubio…` makes an unreadable citation key.
|
|
45
|
+
assert _site_author("karpathy.github.io") == "Karpathy"
|
|
46
|
+
assert _site_author("www.distill.pub") == "Distill"
|
|
47
|
+
assert _site_author("docs.python.org") == "Python"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_a_dateless_page_takes_the_year_from_its_own_url():
|
|
51
|
+
assert _year_in_path("https://karpathy.github.io/2015/05/21/rnn-effectiveness/") == "2015"
|
|
52
|
+
assert _year_in_path("https://example.org/blog/2016/misread-tsne/") == "2016"
|
|
53
|
+
# Not from a query string, where any number at all can appear.
|
|
54
|
+
assert _year_in_path("https://example.org/page?id=2019") == ""
|
|
55
|
+
assert _year_in_path("https://example.org/about") == ""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_the_title_loses_the_site_name_templates_append():
|
|
59
|
+
assert _title_tag("<title>How to Use t-SNE | Distill</title>") == "How to Use t-SNE"
|
|
60
|
+
assert _title_tag("<title>Plain Title</title>") == "Plain Title"
|
|
61
|
+
# A title that is only a site name must survive rather than become empty.
|
|
62
|
+
assert _title_tag("<title>Distill</title>") == "Distill"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def test_meta_is_read_in_either_attribute_order():
|
|
66
|
+
assert _meta_content('<meta property="og:title" content="Hello">', "og:title") == "Hello"
|
|
67
|
+
assert _meta_content('<meta content="Hello" name="og:title">', "og:title") == "Hello"
|
|
68
|
+
assert _meta_content('<meta property="og:title" content="A & B">', "og:title") == "A & B"
|
|
69
|
+
assert _meta_content("<html></html>", "og:title") == ""
|
|
@@ -18,7 +18,7 @@ wheels = [
|
|
|
18
18
|
|
|
19
19
|
[[package]]
|
|
20
20
|
name = "bibcite-cli"
|
|
21
|
-
version = "0.6.
|
|
21
|
+
version = "0.6.2"
|
|
22
22
|
source = { editable = "." }
|
|
23
23
|
dependencies = [
|
|
24
24
|
{ name = "bibtexparser" },
|
|
@@ -75,7 +75,7 @@ name = "exceptiongroup"
|
|
|
75
75
|
version = "1.3.1"
|
|
76
76
|
source = { registry = "https://pypi.org/simple" }
|
|
77
77
|
dependencies = [
|
|
78
|
-
{ name = "typing-extensions"
|
|
78
|
+
{ name = "typing-extensions" },
|
|
79
79
|
]
|
|
80
80
|
sdist = { url = "https://files.pythonhosted.org/packages/50/79/66800aadf48771f6b62f7eb014e352e5d06856655206165d775e675a02c9/exceptiongroup-1.3.1.tar.gz", hash = "sha256:8b412432c6055b0b7d14c310000ae93352ed6754f70fa8f7c34141f91c4e3219", size = 30371, upload-time = "2025-11-21T23:01:54.787Z" }
|
|
81
81
|
wheels = [
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|