bibcite-cli 0.6.1__tar.gz → 0.6.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/PKG-INFO +4 -3
  2. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/Readme.md +3 -2
  3. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/pyproject.toml +1 -1
  4. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/resolve.py +45 -0
  5. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/sources.py +139 -0
  6. bibcite_cli-0.6.2/tests/test_webpages.py +69 -0
  7. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/uv.lock +1 -1
  8. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/.github/workflows/ci.yml +0 -0
  9. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/.github/workflows/publish.yml +0 -0
  10. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/.gitignore +0 -0
  11. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/LICENSE +0 -0
  12. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/assets/bibcite.svg +0 -0
  13. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/skills/bibcite/SKILL.md +0 -0
  14. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/__init__.py +0 -0
  15. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/bibfile.py +0 -0
  16. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/cache.py +0 -0
  17. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/cli.py +0 -0
  18. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/data/strings.bib +0 -0
  19. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/normalize.py +0 -0
  20. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/src/bibcite/venues.py +0 -0
  21. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_bibfile.py +0 -0
  22. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_bugfixes.py +0 -0
  23. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_cli_status.py +0 -0
  24. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_entry_types.py +0 -0
  25. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_normalize.py +0 -0
  26. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_round2.py +0 -0
  27. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_round3.py +0 -0
  28. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_source_retries.py +0 -0
  29. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_status_semantics.py +0 -0
  30. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_strings_override.py +0 -0
  31. {bibcite_cli-0.6.1 → bibcite_cli-0.6.2}/tests/test_venues.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bibcite-cli
3
- Version: 0.6.1
3
+ Version: 0.6.2
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -18,7 +18,7 @@ Description-Content-Type: text/markdown
18
18
  <h1 align="center">bibcite</h1>
19
19
 
20
20
  <p align="center">
21
- Turn an arXiv ID, DOI, or paper title into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
21
+ Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
22
22
  </p>
23
23
 
24
24
  <p align="center">
@@ -105,7 +105,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
105
105
 
106
106
  ## What it handles
107
107
 
108
- - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and paper titles.
108
+ - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
109
+ - A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
109
110
  - It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
110
111
  - It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
111
112
  - It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
@@ -5,7 +5,7 @@
5
5
  <h1 align="center">bibcite</h1>
6
6
 
7
7
  <p align="center">
8
- Turn an arXiv ID, DOI, or paper title into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
8
+ Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
9
9
  </p>
10
10
 
11
11
  <p align="center">
@@ -92,7 +92,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
92
92
 
93
93
  ## What it handles
94
94
 
95
- - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and paper titles.
95
+ - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
96
+ - A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
96
97
  - It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
97
98
  - It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
98
99
  - It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.6.1"
3
+ version = "0.6.2"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -22,6 +22,9 @@ from .sources import (
22
22
  Match,
23
23
  arxiv_metadata,
24
24
  crossref_by_doi,
25
+ PageUnreadable,
26
+ WebPage,
27
+ fetch_web_page,
25
28
  find_published,
26
29
  )
27
30
  from .venues import canonicalize
@@ -43,6 +46,7 @@ ARXIV_URL = re.compile(
43
46
  re.I,
44
47
  )
45
48
  DOI_URL = re.compile(r"doi\.org/(10\.\S+)", re.I)
49
+ WEB_URL = re.compile(r"^https?://\S+$", re.I)
46
50
  DOI_RE = re.compile(r"^10\.\d{4,9}/\S+$")
47
51
  ARXIV_DOI = re.compile(
48
52
  r"^10\.48550/arxiv\."
@@ -74,6 +78,12 @@ def classify(query: str) -> tuple[str, str]:
74
78
  return "arxiv", m.group(1)
75
79
  if DOI_RE.match(q):
76
80
  return "doi", q
81
+ # A URL that is not arXiv and not a DOI is a page: a blog post, a standard,
82
+ # a documentation page. Nothing indexes those, so searching for it as a
83
+ # title only ever returns "no match anywhere" — the page itself is the
84
+ # source.
85
+ if WEB_URL.match(q):
86
+ return "webpage", q
77
87
  return "title", query.strip()
78
88
 
79
89
 
@@ -198,6 +208,31 @@ def _arxiv_only_entry(meta: ArxivMeta) -> dict:
198
208
  }
199
209
 
200
210
 
211
+ def _web_entry(page: WebPage) -> dict:
212
+ """A page cited as @misc, the one type every conference .bst understands.
213
+
214
+ @online is biblatex-only; a NeurIPS or IEEE style would drop the entry
215
+ entirely. howpublished carries the link for the same reason the arXiv
216
+ preprint entry uses it: classic styles print it and ignore `url`. No access
217
+ date, because the tidy step omits `note` from every entry in the file.
218
+ """
219
+ entry = {
220
+ "ENTRYTYPE": "misc",
221
+ "title": page.title,
222
+ "howpublished": f"\\url{{{page.url}}}",
223
+ "url": page.url,
224
+ }
225
+ if page.authors:
226
+ entry["author"] = " and ".join(page.authors)
227
+ elif page.site:
228
+ # No byline: the site is the closest thing to a corporate author, and a
229
+ # key of `anonymousXXXX…` is worse than one naming where it came from.
230
+ entry["author"] = f"{{{page.site}}}"
231
+ if page.year:
232
+ entry["year"] = page.year
233
+ return entry
234
+
235
+
201
236
  def resolve(query: str, require_published: bool = False) -> Resolved:
202
237
  kind, value = classify(query)
203
238
  _log(f"[bibcite] query understood as {kind}: {value}")
@@ -248,6 +283,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
248
283
  entry = _arxiv_only_entry(meta)
249
284
  return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
250
285
 
286
+ if kind == "webpage":
287
+ try:
288
+ page = fetch_web_page(value)
289
+ except PageUnreadable as e:
290
+ raise NotFound(f"could not read {value}: {e}") from e
291
+ if not page.title:
292
+ raise NotFound(f"No title found on the page: {value}")
293
+ _log(f"[web] {page.title}" + (f" ({page.year})" if page.year else ""))
294
+ return Resolved(_finalize(_web_entry(page), None), "webpage", page.site, False)
295
+
251
296
  if kind == "doi":
252
297
  match = crossref_by_doi(value)
253
298
  if not match or not match.title:
@@ -43,6 +43,12 @@ class SourceUnavailable(Exception):
43
43
  """Raised when a source rate-limits/blocks us; the cascade skips it."""
44
44
 
45
45
 
46
+ class PageUnreadable(Exception):
47
+ """A page refused to be read and will refuse again — a different
48
+ identifier, or a hand-written entry, is the next step rather than a
49
+ retry."""
50
+
51
+
46
52
  class TransientSourceError(SourceUnavailable):
47
53
  """Raised after request retries are exhausted without tripping the
48
54
  process-wide circuit breaker for later batch entries."""
@@ -871,3 +877,136 @@ def find_published(
871
877
  if not clean_misses:
872
878
  return None, "unavailable"
873
879
  return None, ("incomplete" if incomplete else "not_found")
880
+
881
+
882
+ @dataclass
883
+ class WebPage:
884
+ """What a web page can say about itself, for citing it as an @misc."""
885
+
886
+ url: str
887
+ title: str
888
+ authors: list[str] = field(default_factory=list)
889
+ year: str = ""
890
+ site: str = ""
891
+
892
+
893
+ def _meta_content(html_text: str, *keys: str) -> str:
894
+ """The content of the first <meta> whose name/property matches a key.
895
+
896
+ Attribute order varies between generators, so both orders are tried rather
897
+ than assuming content comes last.
898
+ """
899
+ for key in keys:
900
+ for pattern in (
901
+ rf'<meta[^>]+(?:name|property)=["\']{re.escape(key)}["\'][^>]*'
902
+ rf'content=["\'](.*?)["\']',
903
+ rf'<meta[^>]+content=["\'](.*?)["\'][^>]*'
904
+ rf'(?:name|property)=["\']{re.escape(key)}["\']',
905
+ ):
906
+ m = re.search(pattern, html_text, re.I | re.S)
907
+ if m and m.group(1).strip():
908
+ return html.unescape(m.group(1).strip())
909
+ return ""
910
+
911
+
912
+ def fetch_web_page(url: str) -> WebPage:
913
+ """Read a page's own description of itself.
914
+
915
+ Blogs, documentation and standards pages are cited constantly and are in
916
+ none of the academic indexes, so there is nothing to look them up in — the
917
+ page itself is the only source. Highwire and Dublin Core tags come first
918
+ because sites that carry them mean them; Open Graph and <title> are the
919
+ fallback every site has.
920
+ """
921
+ try:
922
+ with _client(browser=True) as client:
923
+ response = client.get(url)
924
+ response.raise_for_status()
925
+ body = response.text[:400_000]
926
+ except httpx.HTTPStatusError as e:
927
+ status = e.response.status_code
928
+ # 403 and 404 are settled answers: the page will not become readable on
929
+ # a retry, so say so rather than sending the caller back to wait.
930
+ if status in (401, 403, 404, 410) or 400 <= status < 500:
931
+ raise PageUnreadable(
932
+ f"the page answered {status} — cite it with --bibtex, or use its DOI if it has one"
933
+ ) from e
934
+ raise SourceUnavailable(f"could not fetch {url}: {e}") from e
935
+ except httpx.HTTPError as e:
936
+ raise SourceUnavailable(f"could not fetch {url}: {e}") from e
937
+
938
+ title = (
939
+ _meta_content(body, "citation_title", "DC.title", "og:title", "twitter:title")
940
+ or _title_tag(body)
941
+ )
942
+ authors = [
943
+ author
944
+ for author in (
945
+ _meta_content(body, "citation_author", "DC.creator", "author", "article:author"),
946
+ )
947
+ if author
948
+ ]
949
+ date = _meta_content(
950
+ body,
951
+ "citation_publication_date",
952
+ "citation_date",
953
+ "DC.date",
954
+ "article:published_time",
955
+ "og:updated_time",
956
+ "date",
957
+ )
958
+ year = _year_in(date) or _year_in_path(url)
959
+ site = _meta_content(body, "og:site_name") or _site_author(_host(url))
960
+ return WebPage(url=str(response.url), title=title, authors=authors, year=year, site=site)
961
+
962
+
963
+ def _year_in(text: str) -> str:
964
+ m = re.search(r"(?:19|20)\d{2}", text)
965
+ return m.group(0) if m else ""
966
+
967
+
968
+ def _year_in_path(url: str) -> str:
969
+ """The year a dateless page puts in its own URL.
970
+
971
+ Blog engines write `/2015/05/21/title`, and a citation key of
972
+ `karpathyXXXXunreasonable` is worse than one carrying the year the post
973
+ announces about itself. Only the path is read: a query string can hold any
974
+ number at all.
975
+ """
976
+ path = re.sub(r"^https?://[^/]+", "", url).split("?")[0].split("#")[0]
977
+ for segment in path.split("/"):
978
+ if re.fullmatch(r"(?:19|20)\d{2}", segment):
979
+ return segment
980
+ return ""
981
+
982
+
983
+ def _title_tag(html_text: str) -> str:
984
+ m = re.search(r"<title[^>]*>(.*?)</title>", html_text, re.I | re.S)
985
+ if not m:
986
+ return ""
987
+ # Strip the trailing " — Site Name" many templates append; the site name is
988
+ # recorded separately, and repeating it in the title reads badly in a
989
+ # bibliography.
990
+ title = html.unescape(re.sub(r"\s+", " ", m.group(1))).strip()
991
+ return re.sub(r"\s*[|·—–-]\s*[^|·—–-]{1,40}$", "", title).strip() or title
992
+
993
+
994
+ def _host(url: str) -> str:
995
+ m = re.match(r"https?://(?:www\.)?([^/:]+)", url, re.I)
996
+ return m.group(1) if m else ""
997
+
998
+
999
+ def _site_author(host: str) -> str:
1000
+ """A readable stand-in author for a page with no byline.
1001
+
1002
+ The bare host makes an unreadable key — `karpathygithubioXXXXunreasonable`
1003
+ — so the hosting suffix goes and the name that identifies the site stays:
1004
+ `karpathy.github.io` reads as Karpathy, `docs.python.org` as Python.
1005
+ """
1006
+ labels = [label for label in host.lower().split(".") if label]
1007
+ if not labels:
1008
+ return host
1009
+ generic = {"github", "io", "com", "org", "net", "edu", "gov", "ai", "dev",
1010
+ "co", "uk", "cn", "blog", "www", "docs", "pages", "medium"}
1011
+ named = [label for label in labels if label not in generic]
1012
+ return (named[0] if named else labels[0]).capitalize()
@@ -0,0 +1,69 @@
1
+ """Citing a page that no index knows about.
2
+
3
+ Blogs, documentation and standards pages are cited constantly and appear in
4
+ none of the academic sources, so the page itself has to be the source.
5
+ """
6
+
7
+ from bibcite.resolve import _web_entry, classify
8
+ from bibcite.sources import WebPage, _meta_content, _site_author, _title_tag, _year_in_path
9
+
10
+
11
+ def test_a_plain_url_is_a_webpage_and_the_others_still_are_not():
12
+ assert classify("https://karpathy.github.io/2015/05/21/rnn-effectiveness/")[0] == "webpage"
13
+ assert classify("http://example.edu/notes")[0] == "webpage"
14
+ # The identifiers that resolve properly must keep their own paths.
15
+ assert classify("https://arxiv.org/abs/1706.03762") == ("arxiv", "1706.03762")
16
+ assert classify("https://doi.org/10.1038/s41586-021-03819-2")[0] == "doi"
17
+ assert classify("10.1109/CVPR.2016.90")[0] == "doi"
18
+ assert classify("Attention Is All You Need")[0] == "title"
19
+
20
+
21
+ def test_entry_is_misc_so_conference_styles_can_print_it():
22
+ # @online is biblatex-only: a NeurIPS or IEEE .bst drops the entry.
23
+ entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", year="2024"))
24
+ assert entry["ENTRYTYPE"] == "misc"
25
+ assert entry["howpublished"] == r"\url{https://example.org/post}"
26
+ assert entry["url"] == "https://example.org/post"
27
+ assert entry["year"] == "2024"
28
+
29
+
30
+ def test_a_page_with_no_byline_is_attributed_to_its_site():
31
+ entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", site="Example"))
32
+ # Braced, so the .bst treats it as a corporate name and does not invert it.
33
+ assert entry["author"] == "{Example}"
34
+
35
+
36
+ def test_a_byline_wins_over_the_site():
37
+ entry = _web_entry(
38
+ WebPage(url="https://example.org/post", title="A Post", authors=["Ada Lovelace"], site="Example")
39
+ )
40
+ assert entry["author"] == "Ada Lovelace"
41
+
42
+
43
+ def test_the_site_author_drops_hosting_suffixes():
44
+ # `karpathygithubio…` makes an unreadable citation key.
45
+ assert _site_author("karpathy.github.io") == "Karpathy"
46
+ assert _site_author("www.distill.pub") == "Distill"
47
+ assert _site_author("docs.python.org") == "Python"
48
+
49
+
50
+ def test_a_dateless_page_takes_the_year_from_its_own_url():
51
+ assert _year_in_path("https://karpathy.github.io/2015/05/21/rnn-effectiveness/") == "2015"
52
+ assert _year_in_path("https://example.org/blog/2016/misread-tsne/") == "2016"
53
+ # Not from a query string, where any number at all can appear.
54
+ assert _year_in_path("https://example.org/page?id=2019") == ""
55
+ assert _year_in_path("https://example.org/about") == ""
56
+
57
+
58
+ def test_the_title_loses_the_site_name_templates_append():
59
+ assert _title_tag("<title>How to Use t-SNE | Distill</title>") == "How to Use t-SNE"
60
+ assert _title_tag("<title>Plain Title</title>") == "Plain Title"
61
+ # A title that is only a site name must survive rather than become empty.
62
+ assert _title_tag("<title>Distill</title>") == "Distill"
63
+
64
+
65
+ def test_meta_is_read_in_either_attribute_order():
66
+ assert _meta_content('<meta property="og:title" content="Hello">', "og:title") == "Hello"
67
+ assert _meta_content('<meta content="Hello" name="og:title">', "og:title") == "Hello"
68
+ assert _meta_content('<meta property="og:title" content="A &amp; B">', "og:title") == "A & B"
69
+ assert _meta_content("<html></html>", "og:title") == ""
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.6.1"
21
+ version = "0.6.2"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes