bibcite-cli 0.6.0__tar.gz → 0.6.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/PKG-INFO +8 -3
  2. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/Readme.md +7 -2
  3. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/pyproject.toml +1 -1
  4. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/resolve.py +45 -0
  5. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/sources.py +202 -36
  6. bibcite_cli-0.6.2/tests/test_webpages.py +69 -0
  7. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/uv.lock +2 -2
  8. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/.github/workflows/ci.yml +0 -0
  9. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/.github/workflows/publish.yml +0 -0
  10. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/.gitignore +0 -0
  11. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/LICENSE +0 -0
  12. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/assets/bibcite.svg +0 -0
  13. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/skills/bibcite/SKILL.md +0 -0
  14. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/__init__.py +0 -0
  15. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/bibfile.py +0 -0
  16. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/cache.py +0 -0
  17. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/cli.py +0 -0
  18. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/data/strings.bib +0 -0
  19. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/normalize.py +0 -0
  20. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/src/bibcite/venues.py +0 -0
  21. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_bibfile.py +0 -0
  22. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_bugfixes.py +0 -0
  23. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_cli_status.py +0 -0
  24. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_entry_types.py +0 -0
  25. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_normalize.py +0 -0
  26. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_round2.py +0 -0
  27. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_round3.py +0 -0
  28. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_source_retries.py +0 -0
  29. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_status_semantics.py +0 -0
  30. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_strings_override.py +0 -0
  31. {bibcite_cli-0.6.0 → bibcite_cli-0.6.2}/tests/test_venues.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bibcite-cli
3
- Version: 0.6.0
3
+ Version: 0.6.2
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -18,7 +18,7 @@ Description-Content-Type: text/markdown
18
18
  <h1 align="center">bibcite</h1>
19
19
 
20
20
  <p align="center">
21
- Turn an arXiv ID, DOI, or paper title into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
21
+ Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
22
22
  </p>
23
23
 
24
24
  <p align="center">
@@ -105,7 +105,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
105
105
 
106
106
  ## What it handles
107
107
 
108
- - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and paper titles.
108
+ - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
109
+ - A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
109
110
  - It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
110
111
  - It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
111
112
  - It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
@@ -208,6 +209,10 @@ Install the checkout as an editable command while developing:
208
209
  uv tool install --editable .
209
210
  ```
210
211
 
212
+ ## Acknowledgements
213
+
214
+ Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
215
+
211
216
  ## License
212
217
 
213
218
  `bibcite` is available under the [MIT License](LICENSE).
@@ -5,7 +5,7 @@
5
5
  <h1 align="center">bibcite</h1>
6
6
 
7
7
  <p align="center">
8
- Turn an arXiv ID, DOI, or paper title into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
8
+ Turn an arXiv ID, DOI, paper title, or web page into clean BibTeX, then keep the whole bibliography normalized and deduplicated.
9
9
  </p>
10
10
 
11
11
  <p align="center">
@@ -92,7 +92,8 @@ uvx --from bibcite-cli bibcite get "Attention is all you need"
92
92
 
93
93
  ## What it handles
94
94
 
95
- - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, and paper titles.
95
+ - It accepts arXiv IDs and URLs, arXiv DOIs such as `10.48550/arXiv.1706.03762`, standard DOIs, paper titles, and the URL of a web page.
96
+ - A web page is cited from what the page says about itself, because no index carries blog posts, documentation, or standards. The entry is `@misc` with `howpublished = {\url{...}}`, which every conference `.bst` prints; `@online` is biblatex-only and would be dropped. A page with no byline is attributed to its site, and a page with no date takes the year from its own URL when it has one there.
96
97
  - It searches for a published version before falling back to an arXiv preprint, and it reports when source outages make that check incomplete.
97
98
  - It canonicalizes journal, conference, and workshop names against the bundled venue table, including year-sensitive names such as NIPS and NeurIPS.
98
99
  - It assigns the correct BibTeX entry type and field, such as `@inproceedings` with `booktitle` or `@article` with `journal`.
@@ -195,6 +196,10 @@ Install the checkout as an editable command while developing:
195
196
  uv tool install --editable .
196
197
  ```
197
198
 
199
+ ## Acknowledgements
200
+
201
+ Inspired by [PaperMemory](https://github.com/vict0rsch/PaperMemory), with formatting by [bibtex-tidy](https://github.com/FlamingTempura/bibtex-tidy) and metadata from arXiv, DBLP, Semantic Scholar, Google Scholar, Crossref, Unpaywall, and OpenAlex.
202
+
198
203
  ## License
199
204
 
200
205
  `bibcite` is available under the [MIT License](LICENSE).
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.6.0"
3
+ version = "0.6.2"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -22,6 +22,9 @@ from .sources import (
22
22
  Match,
23
23
  arxiv_metadata,
24
24
  crossref_by_doi,
25
+ PageUnreadable,
26
+ WebPage,
27
+ fetch_web_page,
25
28
  find_published,
26
29
  )
27
30
  from .venues import canonicalize
@@ -43,6 +46,7 @@ ARXIV_URL = re.compile(
43
46
  re.I,
44
47
  )
45
48
  DOI_URL = re.compile(r"doi\.org/(10\.\S+)", re.I)
49
+ WEB_URL = re.compile(r"^https?://\S+$", re.I)
46
50
  DOI_RE = re.compile(r"^10\.\d{4,9}/\S+$")
47
51
  ARXIV_DOI = re.compile(
48
52
  r"^10\.48550/arxiv\."
@@ -74,6 +78,12 @@ def classify(query: str) -> tuple[str, str]:
74
78
  return "arxiv", m.group(1)
75
79
  if DOI_RE.match(q):
76
80
  return "doi", q
81
+ # A URL that is not arXiv and not a DOI is a page: a blog post, a standard,
82
+ # a documentation page. Nothing indexes those, so searching for it as a
83
+ # title only ever returns "no match anywhere" — the page itself is the
84
+ # source.
85
+ if WEB_URL.match(q):
86
+ return "webpage", q
77
87
  return "title", query.strip()
78
88
 
79
89
 
@@ -198,6 +208,31 @@ def _arxiv_only_entry(meta: ArxivMeta) -> dict:
198
208
  }
199
209
 
200
210
 
211
+ def _web_entry(page: WebPage) -> dict:
212
+ """A page cited as @misc, the one type every conference .bst understands.
213
+
214
+ @online is biblatex-only; a NeurIPS or IEEE style would drop the entry
215
+ entirely. howpublished carries the link for the same reason the arXiv
216
+ preprint entry uses it: classic styles print it and ignore `url`. No access
217
+ date, because the tidy step omits `note` from every entry in the file.
218
+ """
219
+ entry = {
220
+ "ENTRYTYPE": "misc",
221
+ "title": page.title,
222
+ "howpublished": f"\\url{{{page.url}}}",
223
+ "url": page.url,
224
+ }
225
+ if page.authors:
226
+ entry["author"] = " and ".join(page.authors)
227
+ elif page.site:
228
+ # No byline: the site is the closest thing to a corporate author, and a
229
+ # key of `anonymousXXXX…` is worse than one naming where it came from.
230
+ entry["author"] = f"{{{page.site}}}"
231
+ if page.year:
232
+ entry["year"] = page.year
233
+ return entry
234
+
235
+
201
236
  def resolve(query: str, require_published: bool = False) -> Resolved:
202
237
  kind, value = classify(query)
203
238
  _log(f"[bibcite] query understood as {kind}: {value}")
@@ -248,6 +283,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
248
283
  entry = _arxiv_only_entry(meta)
249
284
  return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
250
285
 
286
+ if kind == "webpage":
287
+ try:
288
+ page = fetch_web_page(value)
289
+ except PageUnreadable as e:
290
+ raise NotFound(f"could not read {value}: {e}") from e
291
+ if not page.title:
292
+ raise NotFound(f"No title found on the page: {value}")
293
+ _log(f"[web] {page.title}" + (f" ({page.year})" if page.year else ""))
294
+ return Resolved(_finalize(_web_entry(page), None), "webpage", page.site, False)
295
+
251
296
  if kind == "doi":
252
297
  match = crossref_by_doi(value)
253
298
  if not match or not match.title:
@@ -12,6 +12,7 @@ import re
12
12
  import sys
13
13
  import time
14
14
  import xml.etree.ElementTree as ET
15
+ from concurrent.futures import ThreadPoolExecutor, as_completed
15
16
  from dataclasses import dataclass, field
16
17
 
17
18
  import httpx
@@ -23,7 +24,12 @@ BROWSER_UA = (
23
24
  "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
24
25
  "(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
25
26
  )
26
- TIMEOUT = 20.0
27
+ # Per-request cap for the publication-matching sources. Kept short so one slow
28
+ # or half-down source (DBLP in particular) can't stall an interactive resolve;
29
+ # a source that can't answer in time is treated as unavailable → "incomplete",
30
+ # never a false "not published". The arXiv metadata fetch sets its own longer
31
+ # timeout on the request itself, so this does not affect it.
32
+ TIMEOUT = 8.0
27
33
 
28
34
  PREPRINT_VENUES = re.compile(r"arxiv|corr|biorxiv|medrxiv|chemrxiv|ssrn|preprint", re.I)
29
35
  ARXIV_DOI = re.compile(r"^10\.48550/", re.I)
@@ -37,6 +43,12 @@ class SourceUnavailable(Exception):
37
43
  """Raised when a source rate-limits/blocks us; the cascade skips it."""
38
44
 
39
45
 
46
+ class PageUnreadable(Exception):
47
+ """A page refused to be read and will refuse again — a different
48
+ identifier, or a hand-written entry, is the next step rather than a
49
+ retry."""
50
+
51
+
40
52
  class TransientSourceError(SourceUnavailable):
41
53
  """Raised after request retries are exhausted without tripping the
42
54
  process-wide circuit breaker for later batch entries."""
@@ -183,30 +195,32 @@ def _paced_get(
183
195
  params: dict | None = None,
184
196
  headers: dict | None = None,
185
197
  ) -> httpx.Response:
186
- for attempt in range(3):
198
+ for attempt in range(2):
187
199
  wait = min_interval - (time.monotonic() - _LAST_REQUEST.get(source, 0.0))
188
200
  if wait > 0:
189
201
  time.sleep(wait)
190
202
  _LAST_REQUEST[source] = time.monotonic()
191
203
  try:
192
204
  r = c.get(url, params=params, headers=headers)
193
- except httpx.HTTPError as e: # Retry transport errors before failing this entry.
194
- if attempt < 2:
195
- time.sleep(5 * (attempt + 1))
205
+ except httpx.HTTPError as e: # Retry transport errors once before failing.
206
+ if attempt < 1:
207
+ time.sleep(1)
196
208
  continue
197
209
  raise TransientSourceError(
198
210
  f"{source} unreachable ({type(e).__name__})"
199
211
  ) from e
200
212
  if r.status_code == 429:
213
+ # A 429 is usually a persistent rate-limit (e.g. the shared
214
+ # unauthenticated Semantic Scholar pool), not a transient blip, so a
215
+ # long client-side backoff rarely clears it and just stalls an
216
+ # interactive resolve. Take at most one quick retry when the server
217
+ # asks for a short wait, then give up and let the circuit breaker
218
+ # skip this source for the rest of the run.
201
219
  retry_after = int(r.headers.get("Retry-After") or 0)
202
- if retry_after > 30:
203
- raise SourceUnavailable(f"{source} rate-limited (Retry-After {retry_after}s)")
204
- if attempt < 2:
205
- delay = max(retry_after, 4 * (attempt + 1))
206
- _log(f"[{source}] 429 — backing off {delay}s")
207
- time.sleep(delay)
220
+ if attempt < 1 and retry_after <= 2:
221
+ time.sleep(max(retry_after, 1))
208
222
  continue
209
- raise SourceUnavailable(f"{source} rate-limited (429) after backoff retries")
223
+ raise SourceUnavailable(f"{source} rate-limited (429)")
210
224
  return r
211
225
  raise SourceUnavailable(f"{source} unavailable")
212
226
 
@@ -795,32 +809,51 @@ def find_published(
795
809
  _log(f"[cache] hit: {cached.get('venue', '')} ({cached.get('source', '')})")
796
810
  return Match(**cached), "found"
797
811
 
798
- clean_misses = 0
799
812
  # Core sources lost earlier in this run taint this query's verdict too.
800
813
  incomplete = any(n in CORE_SOURCES for n in _DISABLED)
801
- for name, fn in CASCADE:
802
- if name in _DISABLED:
803
- continue
804
- try:
805
- m = fn(title, year, arxiv_id, author_hint)
806
- if m:
807
- cache.put(cache_key, m.__dict__)
808
- return m, "found"
809
- clean_misses += 1
810
- _log(f"[{name}] no publication found")
811
- except TransientSourceError as e:
812
- incomplete = incomplete or name in CORE_SOURCES
813
- _log(
814
- f"[{name}] transient failure for this entry; "
815
- f"will retry on the next: {e}"
816
- )
817
- except SourceUnavailable as e:
818
- _DISABLED[name] = str(e)
819
- incomplete = incomplete or name in CORE_SOURCES
820
- _log(f"[{name}] disabled for the rest of this run: {e}")
821
- except Exception as e: # network hiccup on one source must not kill the run
822
- incomplete = incomplete or name in CORE_SOURCES
823
- _log(f"[{name}] error: {type(e).__name__}: {e}")
814
+ # Query every still-viable source concurrently. A preprint with no published
815
+ # version (the common case) misses everywhere, and used to pay the *sum* of
816
+ # each source's latency; now the wall-clock is the slowest single source.
817
+ # The first verified hit by CASCADE priority still wins.
818
+ active = [(name, fn) for name, fn in CASCADE if name not in _DISABLED]
819
+ outcomes: dict[str, tuple] = {}
820
+ if active:
821
+ with ThreadPoolExecutor(max_workers=len(active)) as pool:
822
+ futures = {
823
+ pool.submit(fn, title, year, arxiv_id, author_hint): name
824
+ for name, fn in active
825
+ }
826
+ for future in as_completed(futures):
827
+ name = futures[future]
828
+ try:
829
+ m = future.result()
830
+ if m:
831
+ outcomes[name] = ("found", m)
832
+ else:
833
+ outcomes[name] = ("miss", None)
834
+ _log(f"[{name}] no publication found")
835
+ except TransientSourceError as e:
836
+ outcomes[name] = ("fail", name in CORE_SOURCES)
837
+ _log(f"[{name}] transient failure for this entry: {e}")
838
+ except SourceUnavailable as e:
839
+ _DISABLED[name] = str(e)
840
+ outcomes[name] = ("fail", name in CORE_SOURCES)
841
+ _log(f"[{name}] disabled for the rest of this run: {e}")
842
+ except Exception as e: # a hiccup on one source must not kill the run
843
+ outcomes[name] = ("fail", name in CORE_SOURCES)
844
+ _log(f"[{name}] error: {type(e).__name__}: {e}")
845
+
846
+ # Prefer the highest-priority source that verified a match.
847
+ for name, _ in CASCADE:
848
+ outcome = outcomes.get(name)
849
+ if outcome and outcome[0] == "found":
850
+ cache.put(cache_key, outcome[1].__dict__)
851
+ return outcome[1], "found"
852
+
853
+ clean_misses = sum(1 for outcome in outcomes.values() if outcome[0] == "miss")
854
+ incomplete = incomplete or any(
855
+ outcome[0] == "fail" and outcome[1] for outcome in outcomes.values()
856
+ )
824
857
 
825
858
  # Exact-title search missed everywhere. Before concluding "no published
826
859
  # version", try the title-drift fallback — camera-ready titles frequently
@@ -844,3 +877,136 @@ def find_published(
844
877
  if not clean_misses:
845
878
  return None, "unavailable"
846
879
  return None, ("incomplete" if incomplete else "not_found")
880
+
881
+
882
+ @dataclass
883
+ class WebPage:
884
+ """What a web page can say about itself, for citing it as an @misc."""
885
+
886
+ url: str
887
+ title: str
888
+ authors: list[str] = field(default_factory=list)
889
+ year: str = ""
890
+ site: str = ""
891
+
892
+
893
+ def _meta_content(html_text: str, *keys: str) -> str:
894
+ """The content of the first <meta> whose name/property matches a key.
895
+
896
+ Attribute order varies between generators, so both orders are tried rather
897
+ than assuming content comes last.
898
+ """
899
+ for key in keys:
900
+ for pattern in (
901
+ rf'<meta[^>]+(?:name|property)=["\']{re.escape(key)}["\'][^>]*'
902
+ rf'content=["\'](.*?)["\']',
903
+ rf'<meta[^>]+content=["\'](.*?)["\'][^>]*'
904
+ rf'(?:name|property)=["\']{re.escape(key)}["\']',
905
+ ):
906
+ m = re.search(pattern, html_text, re.I | re.S)
907
+ if m and m.group(1).strip():
908
+ return html.unescape(m.group(1).strip())
909
+ return ""
910
+
911
+
912
+ def fetch_web_page(url: str) -> WebPage:
913
+ """Read a page's own description of itself.
914
+
915
+ Blogs, documentation and standards pages are cited constantly and are in
916
+ none of the academic indexes, so there is nothing to look them up in — the
917
+ page itself is the only source. Highwire and Dublin Core tags come first
918
+ because sites that carry them mean them; Open Graph and <title> are the
919
+ fallback every site has.
920
+ """
921
+ try:
922
+ with _client(browser=True) as client:
923
+ response = client.get(url)
924
+ response.raise_for_status()
925
+ body = response.text[:400_000]
926
+ except httpx.HTTPStatusError as e:
927
+ status = e.response.status_code
928
+ # 403 and 404 are settled answers: the page will not become readable on
929
+ # a retry, so say so rather than sending the caller back to wait.
930
+ if status in (401, 403, 404, 410) or 400 <= status < 500:
931
+ raise PageUnreadable(
932
+ f"the page answered {status} — cite it with --bibtex, or use its DOI if it has one"
933
+ ) from e
934
+ raise SourceUnavailable(f"could not fetch {url}: {e}") from e
935
+ except httpx.HTTPError as e:
936
+ raise SourceUnavailable(f"could not fetch {url}: {e}") from e
937
+
938
+ title = (
939
+ _meta_content(body, "citation_title", "DC.title", "og:title", "twitter:title")
940
+ or _title_tag(body)
941
+ )
942
+ authors = [
943
+ author
944
+ for author in (
945
+ _meta_content(body, "citation_author", "DC.creator", "author", "article:author"),
946
+ )
947
+ if author
948
+ ]
949
+ date = _meta_content(
950
+ body,
951
+ "citation_publication_date",
952
+ "citation_date",
953
+ "DC.date",
954
+ "article:published_time",
955
+ "og:updated_time",
956
+ "date",
957
+ )
958
+ year = _year_in(date) or _year_in_path(url)
959
+ site = _meta_content(body, "og:site_name") or _site_author(_host(url))
960
+ return WebPage(url=str(response.url), title=title, authors=authors, year=year, site=site)
961
+
962
+
963
+ def _year_in(text: str) -> str:
964
+ m = re.search(r"(?:19|20)\d{2}", text)
965
+ return m.group(0) if m else ""
966
+
967
+
968
+ def _year_in_path(url: str) -> str:
969
+ """The year a dateless page puts in its own URL.
970
+
971
+ Blog engines write `/2015/05/21/title`, and a citation key of
972
+ `karpathyXXXXunreasonable` is worse than one carrying the year the post
973
+ announces about itself. Only the path is read: a query string can hold any
974
+ number at all.
975
+ """
976
+ path = re.sub(r"^https?://[^/]+", "", url).split("?")[0].split("#")[0]
977
+ for segment in path.split("/"):
978
+ if re.fullmatch(r"(?:19|20)\d{2}", segment):
979
+ return segment
980
+ return ""
981
+
982
+
983
+ def _title_tag(html_text: str) -> str:
984
+ m = re.search(r"<title[^>]*>(.*?)</title>", html_text, re.I | re.S)
985
+ if not m:
986
+ return ""
987
+ # Strip the trailing " — Site Name" many templates append; the site name is
988
+ # recorded separately, and repeating it in the title reads badly in a
989
+ # bibliography.
990
+ title = html.unescape(re.sub(r"\s+", " ", m.group(1))).strip()
991
+ return re.sub(r"\s*[|·—–-]\s*[^|·—–-]{1,40}$", "", title).strip() or title
992
+
993
+
994
+ def _host(url: str) -> str:
995
+ m = re.match(r"https?://(?:www\.)?([^/:]+)", url, re.I)
996
+ return m.group(1) if m else ""
997
+
998
+
999
+ def _site_author(host: str) -> str:
1000
+ """A readable stand-in author for a page with no byline.
1001
+
1002
+ The bare host makes an unreadable key — `karpathygithubioXXXXunreasonable`
1003
+ — so the hosting suffix goes and the name that identifies the site stays:
1004
+ `karpathy.github.io` reads as Karpathy, `docs.python.org` as Python.
1005
+ """
1006
+ labels = [label for label in host.lower().split(".") if label]
1007
+ if not labels:
1008
+ return host
1009
+ generic = {"github", "io", "com", "org", "net", "edu", "gov", "ai", "dev",
1010
+ "co", "uk", "cn", "blog", "www", "docs", "pages", "medium"}
1011
+ named = [label for label in labels if label not in generic]
1012
+ return (named[0] if named else labels[0]).capitalize()
@@ -0,0 +1,69 @@
1
+ """Citing a page that no index knows about.
2
+
3
+ Blogs, documentation and standards pages are cited constantly and appear in
4
+ none of the academic sources, so the page itself has to be the source.
5
+ """
6
+
7
+ from bibcite.resolve import _web_entry, classify
8
+ from bibcite.sources import WebPage, _meta_content, _site_author, _title_tag, _year_in_path
9
+
10
+
11
+ def test_a_plain_url_is_a_webpage_and_the_others_still_are_not():
12
+ assert classify("https://karpathy.github.io/2015/05/21/rnn-effectiveness/")[0] == "webpage"
13
+ assert classify("http://example.edu/notes")[0] == "webpage"
14
+ # The identifiers that resolve properly must keep their own paths.
15
+ assert classify("https://arxiv.org/abs/1706.03762") == ("arxiv", "1706.03762")
16
+ assert classify("https://doi.org/10.1038/s41586-021-03819-2")[0] == "doi"
17
+ assert classify("10.1109/CVPR.2016.90")[0] == "doi"
18
+ assert classify("Attention Is All You Need")[0] == "title"
19
+
20
+
21
+ def test_entry_is_misc_so_conference_styles_can_print_it():
22
+ # @online is biblatex-only: a NeurIPS or IEEE .bst drops the entry.
23
+ entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", year="2024"))
24
+ assert entry["ENTRYTYPE"] == "misc"
25
+ assert entry["howpublished"] == r"\url{https://example.org/post}"
26
+ assert entry["url"] == "https://example.org/post"
27
+ assert entry["year"] == "2024"
28
+
29
+
30
+ def test_a_page_with_no_byline_is_attributed_to_its_site():
31
+ entry = _web_entry(WebPage(url="https://example.org/post", title="A Post", site="Example"))
32
+ # Braced, so the .bst treats it as a corporate name and does not invert it.
33
+ assert entry["author"] == "{Example}"
34
+
35
+
36
+ def test_a_byline_wins_over_the_site():
37
+ entry = _web_entry(
38
+ WebPage(url="https://example.org/post", title="A Post", authors=["Ada Lovelace"], site="Example")
39
+ )
40
+ assert entry["author"] == "Ada Lovelace"
41
+
42
+
43
+ def test_the_site_author_drops_hosting_suffixes():
44
+ # `karpathygithubio…` makes an unreadable citation key.
45
+ assert _site_author("karpathy.github.io") == "Karpathy"
46
+ assert _site_author("www.distill.pub") == "Distill"
47
+ assert _site_author("docs.python.org") == "Python"
48
+
49
+
50
+ def test_a_dateless_page_takes_the_year_from_its_own_url():
51
+ assert _year_in_path("https://karpathy.github.io/2015/05/21/rnn-effectiveness/") == "2015"
52
+ assert _year_in_path("https://example.org/blog/2016/misread-tsne/") == "2016"
53
+ # Not from a query string, where any number at all can appear.
54
+ assert _year_in_path("https://example.org/page?id=2019") == ""
55
+ assert _year_in_path("https://example.org/about") == ""
56
+
57
+
58
+ def test_the_title_loses_the_site_name_templates_append():
59
+ assert _title_tag("<title>How to Use t-SNE | Distill</title>") == "How to Use t-SNE"
60
+ assert _title_tag("<title>Plain Title</title>") == "Plain Title"
61
+ # A title that is only a site name must survive rather than become empty.
62
+ assert _title_tag("<title>Distill</title>") == "Distill"
63
+
64
+
65
+ def test_meta_is_read_in_either_attribute_order():
66
+ assert _meta_content('<meta property="og:title" content="Hello">', "og:title") == "Hello"
67
+ assert _meta_content('<meta content="Hello" name="og:title">', "og:title") == "Hello"
68
+ assert _meta_content('<meta property="og:title" content="A &amp; B">', "og:title") == "A & B"
69
+ assert _meta_content("<html></html>", "og:title") == ""
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.6.0"
21
+ version = "0.6.2"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
@@ -75,7 +75,7 @@ name = "exceptiongroup"
75
75
  version = "1.3.1"
76
76
  source = { registry = "https://pypi.org/simple" }
77
77
  dependencies = [
78
- { name = "typing-extensions", marker = "python_full_version < '3.13'" },
78
+ { name = "typing-extensions" },
79
79
  ]
80
80
  sdist = { url = "https://files.pythonhosted.org/packages/50/79/66800aadf48771f6b62f7eb014e352e5d06856655206165d775e675a02c9/exceptiongroup-1.3.1.tar.gz", hash = "sha256:8b412432c6055b0b7d14c310000ae93352ed6754f70fa8f7c34141f91c4e3219", size = 30371, upload-time = "2025-11-21T23:01:54.787Z" }
81
81
  wheels = [
File without changes
File without changes