bibcite-cli 0.6.1__tar.gz → 0.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/PKG-INFO +4 -3
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/Readme.md +3 -2
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/pyproject.toml +1 -1
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/resolve.py +45 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/sources.py +139 -0
- bibcite_cli-0.6.2/tests/test_webpages.py +69 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/uv.lock +1 -1
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/.github/workflows/ci.yml +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/.github/workflows/publish.yml +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/.gitignore +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/LICENSE +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/assets/bibcite.svg +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/skills/bibcite/SKILL.md +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/__init__.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/bibfile.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/cli.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/normalize.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/venues.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_cli_status.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_round2.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_round3.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_source_retries.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_status_semantics.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_venues.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.2
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -18,7 +18,7 @@ Description-Content-Type: text/markdown
|
|
|
18
18
|
<h1 align="center">bibcite</h1>
|
|
19
19
|
|
|
20
20
|
<p align="center">
|
|
21
|
-
Turn an arXiv ID, DOI,
|
|
21
|
+
Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
|
|
22
22
|
</p>
|
|
23
23
|
|
|
24
24
|
<p align="center">
|
|
@@ -105,7 +105,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
|
|
|
105
105
|
|
|
106
106
|
## What it handles
|
|
107
107
|
|
|
108
|
-
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and
|
|
108
|
+
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
|
|
109
|
+
- A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
|
|
109
110
|
- It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
|
|
110
111
|
- It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
|
|
111
112
|
- It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
<h1 align="center">bibcite</h1>
|
|
6
6
|
|
|
7
7
|
<p align="center">
|
|
8
|
-
Turn an arXiv ID, DOI,
|
|
8
|
+
Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
|
|
9
9
|
</p>
|
|
10
10
|
|
|
11
11
|
<p align="center">
|
|
@@ -92,7 +92,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
|
|
|
92
92
|
|
|
93
93
|
## What it handles
|
|
94
94
|
|
|
95
|
-
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and
|
|
95
|
+
- It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
|
|
96
|
+
- A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
|
|
96
97
|
- It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
|
|
97
98
|
- It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
|
|
98
99
|
- It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
|
|
@@ -22,6 +22,9 @@ from .sources import (
|
|
|
22
22
|
Match,
|
|
23
23
|
arxiv_metadata,
|
|
24
24
|
crossref_by_doi,
|
|
25
|
+
PageUnreadable,
|
|
26
|
+
WebPage,
|
|
27
|
+
fetch_web_page,
|
|
25
28
|
find_published,
|
|
26
29
|
)
|
|
27
30
|
from .venues import canonicalize
|
|
@@ -43,6 +46,7 @@ ARXIV_URL = re.compile(
|
|
|
43
46
|
re.I,
|
|
44
47
|
)
|
|
45
48
|
DOI_URL = re.compile(r"doi\.org/(10\.\S+)", re.I)
|
|
49
|
+
WEB_URL = re.compile(r"^https?://\S+$", re.I)
|
|
46
50
|
DOI_RE = re.compile(r"^10\.\d{4,9}/\S+$")
|
|
47
51
|
ARXIV_DOI = re.compile(
|
|
48
52
|
r"^10\.48550/arxiv\."
|
|
@@ -74,6 +78,12 @@ def classify(query: str) -> tuple[str, str]:
|
|
|
74
78
|
return "arxiv", m.group(1)
|
|
75
79
|
if DOI_RE.match(q):
|
|
76
80
|
return "doi", q
|
|
81
|
+
# A URL that is not arXiv and not a DOI is a page: a blog post, a standard,
|
|
82
|
+
# a documentation page. Nothing indexes those, so searching for it as a
|
|
83
|
+
# title only ever returns "no match anywhere" — the page itself is the
|
|
84
|
+
# source.
|
|
85
|
+
if WEB_URL.match(q):
|
|
86
|
+
return "webpage", q
|
|
77
87
|
return "title", query.strip()
|
|
78
88
|
|
|
79
89
|
|
|
@@ -198,6 +208,31 @@ def _arxiv_only_entry(meta: ArxivMeta) -> dict:
|
|
|
198
208
|
}
|
|
199
209
|
|
|
200
210
|
|
|
211
|
+
def _web_entry(page: WebPage) -> dict:
|
|
212
|
+
"""A page cited as @misc, the one type every conference .bst understands.
|
|
213
|
+
|
|
214
|
+
@online is biblatex-only; a NeurIPS or IEEE style would drop the entry
|
|
215
|
+
entirely. howpublished carries the link for the same reason the arXiv
|
|
216
|
+
preprint entry uses it: classic styles print it and ignore `url`. No access
|
|
217
|
+
date, because the tidy step omits `note` from every entry in the file.
|
|
218
|
+
"""
|
|
219
|
+
entry = {
|
|
220
|
+
"ENTRYTYPE": "misc",
|
|
221
|
+
"title": page.title,
|
|
222
|
+
"howpublished": f"\\url{{{page.url}}}",
|
|
223
|
+
"url": page.url,
|
|
224
|
+
}
|
|
225
|
+
if page.authors:
|
|
226
|
+
entry["author"] = " and ".join(page.authors)
|
|
227
|
+
elif page.site:
|
|
228
|
+
# No byline: the site is the closest thing to a corporate author, and a
|
|
229
|
+
# key of `anonymousXXXX…` is worse than one naming where it came from.
|
|
230
|
+
entry["author"] = f"{{{page.site}}}"
|
|
231
|
+
if page.year:
|
|
232
|
+
entry["year"] = page.year
|
|
233
|
+
return entry
|
|
234
|
+
|
|
235
|
+
|
|
201
236
|
def resolve(query: str, require_published: bool = False) -> Resolved:
|
|
202
237
|
kind, value = classify(query)
|
|
203
238
|
_log(f"[bibcite] query understood as {kind}: {value}")
|
|
@@ -248,6 +283,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
|
|
|
248
283
|
entry = _arxiv_only_entry(meta)
|
|
249
284
|
return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
|
|
250
285
|
|
|
286
|
+
if kind == "webpage":
|
|
287
|
+
try:
|
|
288
|
+
page = fetch_web_page(value)
|
|
289
|
+
except PageUnreadable as e:
|
|
290
|
+
raise NotFound(f"could not read {value}: {e}") from e
|
|
291
|
+
if not page.title:
|
|
292
|
+
raise NotFound(f"No title found on the page: {value}")
|
|
293
|
+
_log(f"[web] {page.title}" + (f" ({page.year})" if page.year else ""))
|
|
294
|
+
return Resolved(_finalize(_web_entry(page), None), "webpage", page.site, False)
|
|
295
|
+
|
|
251
296
|
if kind == "doi":
|
|
252
297
|
match = crossref_by_doi(value)
|
|
253
298
|
if not match or not match.title:
|
|
@@ -43,6 +43,12 @@ class SourceUnavailable(Exception):
|
|
|
43
43
|
"""Raised when a source rate-limits/blocks us; the cascade skips it."""
|
|
44
44
|
|
|
45
45
|
|
|
46
|
+
class PageUnreadable(Exception):
|
|
47
|
+
"""A page refused to be read and will refuse again — a different
|
|
48
|
+
identifier, or a hand-written entry, is the next step rather than a
|
|
49
|
+
retry."""
|
|
50
|
+
|
|
51
|
+
|
|
46
52
|
class TransientSourceError(SourceUnavailable):
|
|
47
53
|
"""Raised after request retries are exhausted without tripping the
|
|
48
54
|
process-wide circuit breaker for later batch entries."""
|
|
@@ -871,3 +877,136 @@ def find_published(
|
|
|
871
877
|
if not clean_misses:
|
|
872
878
|
return None, "unavailable"
|
|
873
879
|
return None, ("incomplete" if incomplete else "not_found")
|
|
880
|
+
|
|
881
|
+
|
|
882
|
+
@dataclass
|
|
883
|
+
class WebPage:
|
|
884
|
+
"""What a web page can say about itself, for citing it as an @misc."""
|
|
885
|
+
|
|
886
|
+
url: str
|
|
887
|
+
title: str
|
|
888
|
+
authors: list[str] = field(default_factory=list)
|
|
889
|
+
year: str = ""
|
|
890
|
+
site: str = ""
|
|
891
|
+
|
|
892
|
+
|
|
893
|
+
def _meta_content(html_text: str, *keys: str) -> str:
|
|
894
|
+
"""The content of the first <meta> whose name/property matches a key.
|
|
895
|
+
|
|
896
|
+
Attribute order varies between generators, so both orders are tried rather
|
|
897
|
+
than assuming content comes last.
|
|
898
|
+
"""
|
|
899
|
+
for key in keys:
|
|
900
|
+
for pattern in (
|
|
901
|
+
rf'<meta[^>]+(?:name|property)=["\']{re.escape(key)}["\'][^>]*'
|
|
902
|
+
rf'content=["\'](.*?)["\']',
|
|
903
|
+
rf'<meta[^>]+content=["\'](.*?)["\'][^>]*'
|
|
904
|
+
rf'(?:name|property)=["\']{re.escape(key)}["\']',
|
|
905
|
+
):
|
|
906
|
+
m = re.search(pattern, html_text, re.I | re.S)
|
|
907
|
+
if m and m.group(1).strip():
|
|
908
|
+
return html.unescape(m.group(1).strip())
|
|
909
|
+
return ""
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
def fetch_web_page(url: str) -> WebPage:
|
|
913
|
+
"""Read a page's own description of itself.
|
|
914
|
+
|
|
915
|
+
Blogs, documentation and standards pages are cited constantly and are in
|
|
916
|
+
none of the academic indexes, so there is nothing to look them up in — the
|
|
917
|
+
page itself is the only source. Highwire and Dublin Core tags come first
|
|
918
|
+
because sites that carry them mean them; Open Graph and <title> are the
|
|
919
|
+
fallback every site has.
|
|
920
|
+
"""
|
|
921
|
+
try:
|
|
922
|
+
with _client(browser=True) as client:
|
|
923
|
+
response = client.get(url)
|
|
924
|
+
response.raise_for_status()
|
|
925
|
+
body = response.text[:400_000]
|
|
926
|
+
except httpx.HTTPStatusError as e:
|
|
927
|
+
status = e.response.status_code
|
|
928
|
+
# 403 and 404 are settled answers: the page will not become readable on
|
|
929
|
+
# a retry, so say so rather than sending the caller back to wait.
|
|
930
|
+
if status in (401, 403, 404, 410) or 400 <= status < 500:
|
|
931
|
+
raise PageUnreadable(
|
|
932
|
+
f"the page answered {status} — cite it with --bibtex, or use its DOI if it has one"
|
|
933
|
+
) from e
|
|
934
|
+
raise SourceUnavailable(f"could not fetch {url}: {e}") from e
|
|
935
|
+
except httpx.HTTPError as e:
|
|
936
|
+
raise SourceUnavailable(f"could not fetch {url}: {e}") from e
|
|
937
|
+
|
|
938
|
+
title = (
|
|
939
|
+
_meta_content(body, "citation_title", "DC.title", "og:title", "twitter:title")
|
|
940
|
+
or _title_tag(body)
|
|
941
|
+
)
|
|
942
|
+
authors = [
|
|
943
|
+
author
|
|
944
|
+
for author in (
|
|
945
|
+
_meta_content(body, "citation_author", "DC.creator", "author", "article:author"),
|
|
946
|
+
)
|
|
947
|
+
if author
|
|
948
|
+
]
|
|
949
|
+
date = _meta_content(
|
|
950
|
+
body,
|
|
951
|
+
"citation_publication_date",
|
|
952
|
+
"citation_date",
|
|
953
|
+
"DC.date",
|
|
954
|
+
"article:published_time",
|
|
955
|
+
"og:updated_time",
|
|
956
|
+
"date",
|
|
957
|
+
)
|
|
958
|
+
year = _year_in(date) or _year_in_path(url)
|
|
959
|
+
site = _meta_content(body, "og:site_name") or _site_author(_host(url))
|
|
960
|
+
return WebPage(url=str(response.url), title=title, authors=authors, year=year, site=site)
|
|
961
|
+
|
|
962
|
+
|
|
963
|
+
def _year_in(text: str) -> str:
|
|
964
|
+
m = re.search(r"(?:19|20)\d{2}", text)
|
|
965
|
+
return m.group(0) if m else ""
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
def _year_in_path(url: str) -> str:
|
|
969
|
+
"""The year a dateless page puts in its own URL.
|
|
970
|
+
|
|
971
|
+
Blog engines write `/2015/05/21/title`, and a citation key of
|
|
972
|
+
`karpathyXXXXunreasonable` is worse than one carrying the year the post
|
|
973
|
+
announces about itself. Only the path is read: a query string can hold any
|
|
974
|
+
number at all.
|
|
975
|
+
"""
|
|
976
|
+
path = re.sub(r"^https?://[^/]+", "", url).split("?")[0].split("#")[0]
|
|
977
|
+
for segment in path.split("/"):
|
|
978
|
+
if re.fullmatch(r"(?:19|20)\d{2}", segment):
|
|
979
|
+
return segment
|
|
980
|
+
return ""
|
|
981
|
+
|
|
982
|
+
|
|
983
|
+
def _title_tag(html_text: str) -> str:
|
|
984
|
+
m = re.search(r"<title[^>]*>(.*?)</title>", html_text, re.I | re.S)
|
|
985
|
+
if not m:
|
|
986
|
+
return ""
|
|
987
|
+
# Strip the trailing " — Site Name" many templates append; the site name is
|
|
988
|
+
# recorded separately, and repeating it in the title reads badly in a
|
|
989
|
+
# bibliography.
|
|
990
|
+
title = html.unescape(re.sub(r"\s+", " ", m.group(1))).strip()
|
|
991
|
+
return re.sub(r"\s*[|·—–-]\s*[^|·—–-]{1,40}$", "", title).strip() or title
|
|
992
|
+
|
|
993
|
+
|
|
994
|
+
def _host(url: str) -> str:
|
|
995
|
+
m = re.match(r"https?://(?:www\.)?([^/:]+)", url, re.I)
|
|
996
|
+
return m.group(1) if m else ""
|
|
997
|
+
|
|
998
|
+
|
|
999
|
+
def _site_author(host: str) -> str:
|
|
1000
|
+
"""A readable stand-in author for a page with no byline.
|
|
1001
|
+
|
|
1002
|
+
The bare host makes an unreadable key — `karpathygithubioXXXXunreasonable`
|
|
1003
|
+
— so the hosting suffix goes and the name that identifies the site stays:
|
|
1004
|
+
`karpathy.github.io` reads as Karpathy, `docs.python.org` as Python.
|
|
1005
|
+
"""
|
|
1006
|
+
labels = [label for label in host.lower().split(".") if label]
|
|
1007
|
+
if not labels:
|
|
1008
|
+
return host
|
|
1009
|
+
generic = {"github", "io", "com", "org", "net", "edu", "gov", "ai", "dev",
|
|
1010
|
+
"co", "uk", "cn", "blog", "www", "docs", "pages", "medium"}
|
|
1011
|
+
named = [label for label in labels if label not in generic]
|
|
1012
|
+
return (named[0] if named else labels[0]).capitalize()
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Citing a page that no index knows about.
|
|
2
|
+
|
|
3
|
+
Blogs, documentation and standards pages are cited constantly and appear in
|
|
4
|
+
none of the academic sources, so the page itself has to be the source.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from bibcite.resolve import _web_entry, classify
|
|
8
|
+
from bibcite.sources import WebPage, _meta_content, _site_author, _title_tag, _year_in_path
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_a_plain_url_is_a_webpage_and_the_others_still_are_not():
|
|
12
|
+
assert classify("https://karpathy.github.io/2015/05/21/rnn-effectiveness/")[0] == "webpage"
|
|
13
|
+
assert classify("http://example.edu/notes")[0] == "webpage"
|
|
14
|
+
# The identifiers that resolve properly must keep their own paths.
|
|
15
|
+
assert classify("https://arxiv.org/abs/1706.03762") == ("arxiv", "1706.03762")
|
|
16
|
+
assert classify("https://doi.org/10.1038/s41586-021-03819-2")[0] == "doi"
|
|
17
|
+
assert classify("10.1109/CVPR.2016.90")[0] == "doi"
|
|
18
|
+
assert classify("Attention Is All You Need")[0] == "title"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_entry_is_misc_so_conference_styles_can_print_it():
|
|
22
|
+
# @online is biblatex-only: a NeurIPS or IEEE .bst drops the entry.
|
|
23
|
+
entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", year="2024"))
|
|
24
|
+
assert entry["ENTRYTYPE"] == "misc"
|
|
25
|
+
assert entry["howpublished"] == r"\url{https://example.org/post}"
|
|
26
|
+
assert entry["url"] == "https://example.org/post"
|
|
27
|
+
assert entry["year"] == "2024"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_a_page_with_no_byline_is_attributed_to_its_site():
|
|
31
|
+
entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", site="Example"))
|
|
32
|
+
# Braced, so the .bst treats it as a corporate name and does not invert it.
|
|
33
|
+
assert entry["author"] == "{Example}"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_a_byline_wins_over_the_site():
|
|
37
|
+
entry = _web_entry(
|
|
38
|
+
WebPage(url="https://example.org/post", title="A Post", authors=["Ada Lovelace"], site="Example")
|
|
39
|
+
)
|
|
40
|
+
assert entry["author"] == "Ada Lovelace"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def test_the_site_author_drops_hosting_suffixes():
|
|
44
|
+
# `karpathygithubio…` makes an unreadable citation key.
|
|
45
|
+
assert _site_author("karpathy.github.io") == "Karpathy"
|
|
46
|
+
assert _site_author("www.distill.pub") == "Distill"
|
|
47
|
+
assert _site_author("docs.python.org") == "Python"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_a_dateless_page_takes_the_year_from_its_own_url():
|
|
51
|
+
assert _year_in_path("https://karpathy.github.io/2015/05/21/rnn-effectiveness/") == "2015"
|
|
52
|
+
assert _year_in_path("https://example.org/blog/2016/misread-tsne/") == "2016"
|
|
53
|
+
# Not from a query string, where any number at all can appear.
|
|
54
|
+
assert _year_in_path("https://example.org/page?id=2019") == ""
|
|
55
|
+
assert _year_in_path("https://example.org/about") == ""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_the_title_loses_the_site_name_templates_append():
|
|
59
|
+
assert _title_tag("<title>How to Use t-SNE | Distill</title>") == "How to Use t-SNE"
|
|
60
|
+
assert _title_tag("<title>Plain Title</title>") == "Plain Title"
|
|
61
|
+
# A title that is only a site name must survive rather than become empty.
|
|
62
|
+
assert _title_tag("<title>Distill</title>") == "Distill"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def test_meta_is_read_in_either_attribute_order():
|
|
66
|
+
assert _meta_content('<meta property="og:title" content="Hello">', "og:title") == "Hello"
|
|
67
|
+
assert _meta_content('<meta content="Hello" name="og:title">', "og:title") == "Hello"
|
|
68
|
+
assert _meta_content('<meta property="og:title" content="A & B">', "og:title") == "A & B"
|
|
69
|
+
assert _meta_content("<html></html>", "og:title") == ""
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|