bibcite-cli 0.5.2__tar.gz → 0.5.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/PKG-INFO +1 -1
  2. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/pyproject.toml +1 -1
  3. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/__init__.py +1 -1
  4. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/bibfile.py +2 -4
  5. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/sources.py +24 -16
  6. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_round3.py +8 -0
  7. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/uv.lock +1 -1
  8. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/.gitignore +0 -0
  9. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/LICENSE +0 -0
  10. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/Readme.md +0 -0
  11. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/cache.py +0 -0
  12. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/cli.py +0 -0
  13. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/data/strings.bib +0 -0
  14. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/normalize.py +0 -0
  15. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/resolve.py +0 -0
  16. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/venues.py +0 -0
  17. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_bibfile.py +0 -0
  18. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_bugfixes.py +0 -0
  19. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_entry_types.py +0 -0
  20. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_normalize.py +0 -0
  21. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_round2.py +0 -0
  22. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_status_semantics.py +0 -0
  23. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_strings_override.py +0 -0
  24. {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_venues.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bibcite-cli
3
- Version: 0.5.2
3
+ Version: 0.5.4
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.5.2"
3
+ version = "0.5.4"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """bibcite: canonical BibTeX resolution for papers (arXiv id / DOI / title)."""
2
2
 
3
- __version__ = "0.5.2"
3
+ __version__ = "0.5.4"
@@ -20,16 +20,14 @@ from .normalize import norm_title
20
20
  # \cite{} commands valid.
21
21
  TIDY_ARGS = [
22
22
  "--modify",
23
- # volume/number/pages/doi are kept (bibliographic substance the user
24
- # asked to retain); the omit list drops only true noise.
25
- "--omit=publisher,timestamp,biburl,bibsource,abstract,month,series,editor,note,date,address",
23
+ "--omit=pages,publisher,doi,timestamp,biburl,bibsource,abstract,month,series,volume,editor,note,date,number,address,issn,isbn",
26
24
  "--curly",
27
25
  "--blank-lines",
28
26
  "--trailing-commas",
29
27
  "--sort=-year",
30
28
  "--duplicates=citation",
31
29
  "--merge=first",
32
- "--sort-fields=author,title,booktitle,journal,volume,number,pages,year,doi,url,pdf",
30
+ "--sort-fields=author,title,booktitle,journal,year,url,pdf",
33
31
  "--strip-enclosing-braces",
34
32
  "--tidy-comments",
35
33
  ]
@@ -209,6 +209,28 @@ def _dblp_get(c: httpx.Client, url: str, params: dict | None = None) -> httpx.Re
209
209
  return _paced_get(c, url, "dblp", 0.8, params=params)
210
210
 
211
211
 
212
+ def _dblp_sanitize(q: str) -> str:
213
+ """DBLP's search parser 500s deterministically on queries containing its
214
+ syntax characters (':' in subtitled papers, '?' in question titles).
215
+ They tokenize on punctuation anyway, so replacing with spaces loses
216
+ nothing."""
217
+ return re.sub(r"[^\w\s.-]", " ", q)
218
+
219
+
220
+ def _dblp_search(c: httpx.Client, q: str, h: int = 100) -> list:
221
+ """One search request; a 500 (broad all-common-words query × large h
222
+ times out their backend) is retried once with a small result window
223
+ before giving up on this query variant."""
224
+ url = "https://dblp.org/search/publ/api"
225
+ q = _dblp_sanitize(q)
226
+ r = _dblp_get(c, url, params={"q": q, "format": "json", "h": h})
227
+ if r.status_code == 500 and h > 10:
228
+ _log("[dblp] 500 on broad query — retrying with h=10")
229
+ r = _dblp_get(c, url, params={"q": q, "format": "json", "h": 10})
230
+ r.raise_for_status()
231
+ return r.json().get("result", {}).get("hits", {}).get("hit", []) or []
232
+
233
+
212
234
  def try_dblp(title: str, author_hint: str = "") -> Match | None:
213
235
  """DBLP search. Generic titles ("X is all you need") drown in DBLP's
214
236
  ranking, so when we know the first author we query with their last name
@@ -219,15 +241,7 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
219
241
  queries.append(title)
220
242
  with _client() as c:
221
243
  for q in queries:
222
- r = _dblp_get(
223
- c,
224
- "https://dblp.org/search/publ/api",
225
- params={"q": q, "format": "json", "h": 100},
226
- )
227
- r.raise_for_status()
228
- hits = (
229
- r.json().get("result", {}).get("hits", {}).get("hit", []) or []
230
- )
244
+ hits = _dblp_search(c, q)
231
245
  # Earliest year first: prefer the original conference publication
232
246
  # over later journal extensions (same heuristic as PaperMemory).
233
247
  hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
@@ -283,13 +297,7 @@ def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None
283
297
  return None
284
298
  q = " ".join([author_hint] + tokens)
285
299
  with _client() as c:
286
- r = _dblp_get(
287
- c,
288
- "https://dblp.org/search/publ/api",
289
- params={"q": q, "format": "json", "h": 100},
290
- )
291
- r.raise_for_status()
292
- hits = r.json().get("result", {}).get("hits", {}).get("hit", []) or []
300
+ hits = _dblp_search(c, q)
293
301
  hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
294
302
  for hit in hits:
295
303
  info = hit.get("info", {})
@@ -84,3 +84,11 @@ def test_scrub_leaves_clean_files_alone(tmp_path: Path):
84
84
  bib.write_text(original)
85
85
  _scrub_month_strings(bib)
86
86
  assert bib.read_text() == original # untouched, not even rewritten
87
+
88
+
89
+ def test_dblp_sanitize_strips_syntax_chars():
90
+ from bibcite.sources import _dblp_sanitize
91
+
92
+ q = _dblp_sanitize("LeJEPA: Provable? Self-Supervised (Learning)")
93
+ assert ":" not in q and "?" not in q and "(" not in q
94
+ assert "Self-Supervised" in q # hyphens survive
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.5.2"
21
+ version = "0.5.4"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes
File without changes