bibcite-cli 0.5.2__tar.gz → 0.5.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/PKG-INFO +1 -1
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/pyproject.toml +1 -1
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/__init__.py +1 -1
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/bibfile.py +2 -4
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/sources.py +24 -16
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_round3.py +8 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/uv.lock +1 -1
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/.gitignore +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/LICENSE +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/Readme.md +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/cli.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/normalize.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/resolve.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/src/bibcite/venues.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_round2.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_status_semantics.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.5.2 → bibcite_cli-0.5.4}/tests/test_venues.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.4
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -20,16 +20,14 @@ from .normalize import norm_title
|
|
|
20
20
|
# \cite{} commands valid.
|
|
21
21
|
TIDY_ARGS = [
|
|
22
22
|
"--modify",
|
|
23
|
-
|
|
24
|
-
# asked to retain); the omit list drops only true noise.
|
|
25
|
-
"--omit=publisher,timestamp,biburl,bibsource,abstract,month,series,editor,note,date,address",
|
|
23
|
+
"--omit=pages,publisher,doi,timestamp,biburl,bibsource,abstract,month,series,volume,editor,note,date,number,address,issn,isbn",
|
|
26
24
|
"--curly",
|
|
27
25
|
"--blank-lines",
|
|
28
26
|
"--trailing-commas",
|
|
29
27
|
"--sort=-year",
|
|
30
28
|
"--duplicates=citation",
|
|
31
29
|
"--merge=first",
|
|
32
|
-
"--sort-fields=author,title,booktitle,journal,
|
|
30
|
+
"--sort-fields=author,title,booktitle,journal,year,url,pdf",
|
|
33
31
|
"--strip-enclosing-braces",
|
|
34
32
|
"--tidy-comments",
|
|
35
33
|
]
|
|
@@ -209,6 +209,28 @@ def _dblp_get(c: httpx.Client, url: str, params: dict | None = None) -> httpx.Re
|
|
|
209
209
|
return _paced_get(c, url, "dblp", 0.8, params=params)
|
|
210
210
|
|
|
211
211
|
|
|
212
|
+
def _dblp_sanitize(q: str) -> str:
|
|
213
|
+
"""DBLP's search parser 500s deterministically on queries containing its
|
|
214
|
+
syntax characters (':' in subtitled papers, '?' in question titles).
|
|
215
|
+
They tokenize on punctuation anyway, so replacing with spaces loses
|
|
216
|
+
nothing."""
|
|
217
|
+
return re.sub(r"[^\w\s.-]", " ", q)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _dblp_search(c: httpx.Client, q: str, h: int = 100) -> list:
|
|
221
|
+
"""One search request; a 500 (broad all-common-words query × large h
|
|
222
|
+
times out their backend) is retried once with a small result window
|
|
223
|
+
before giving up on this query variant."""
|
|
224
|
+
url = "https://dblp.org/search/publ/api"
|
|
225
|
+
q = _dblp_sanitize(q)
|
|
226
|
+
r = _dblp_get(c, url, params={"q": q, "format": "json", "h": h})
|
|
227
|
+
if r.status_code == 500 and h > 10:
|
|
228
|
+
_log("[dblp] 500 on broad query — retrying with h=10")
|
|
229
|
+
r = _dblp_get(c, url, params={"q": q, "format": "json", "h": 10})
|
|
230
|
+
r.raise_for_status()
|
|
231
|
+
return r.json().get("result", {}).get("hits", {}).get("hit", []) or []
|
|
232
|
+
|
|
233
|
+
|
|
212
234
|
def try_dblp(title: str, author_hint: str = "") -> Match | None:
|
|
213
235
|
"""DBLP search. Generic titles ("X is all you need") drown in DBLP's
|
|
214
236
|
ranking, so when we know the first author we query with their last name
|
|
@@ -219,15 +241,7 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
|
|
|
219
241
|
queries.append(title)
|
|
220
242
|
with _client() as c:
|
|
221
243
|
for q in queries:
|
|
222
|
-
|
|
223
|
-
c,
|
|
224
|
-
"https://dblp.org/search/publ/api",
|
|
225
|
-
params={"q": q, "format": "json", "h": 100},
|
|
226
|
-
)
|
|
227
|
-
r.raise_for_status()
|
|
228
|
-
hits = (
|
|
229
|
-
r.json().get("result", {}).get("hits", {}).get("hit", []) or []
|
|
230
|
-
)
|
|
244
|
+
hits = _dblp_search(c, q)
|
|
231
245
|
# Earliest year first: prefer the original conference publication
|
|
232
246
|
# over later journal extensions (same heuristic as PaperMemory).
|
|
233
247
|
hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
|
|
@@ -283,13 +297,7 @@ def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None
|
|
|
283
297
|
return None
|
|
284
298
|
q = " ".join([author_hint] + tokens)
|
|
285
299
|
with _client() as c:
|
|
286
|
-
|
|
287
|
-
c,
|
|
288
|
-
"https://dblp.org/search/publ/api",
|
|
289
|
-
params={"q": q, "format": "json", "h": 100},
|
|
290
|
-
)
|
|
291
|
-
r.raise_for_status()
|
|
292
|
-
hits = r.json().get("result", {}).get("hits", {}).get("hit", []) or []
|
|
300
|
+
hits = _dblp_search(c, q)
|
|
293
301
|
hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
|
|
294
302
|
for hit in hits:
|
|
295
303
|
info = hit.get("info", {})
|
|
@@ -84,3 +84,11 @@ def test_scrub_leaves_clean_files_alone(tmp_path: Path):
|
|
|
84
84
|
bib.write_text(original)
|
|
85
85
|
_scrub_month_strings(bib)
|
|
86
86
|
assert bib.read_text() == original # untouched, not even rewritten
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_dblp_sanitize_strips_syntax_chars():
|
|
90
|
+
from bibcite.sources import _dblp_sanitize
|
|
91
|
+
|
|
92
|
+
q = _dblp_sanitize("LeJEPA: Provable? Self-Supervised (Learning)")
|
|
93
|
+
assert ":" not in q and "?" not in q and "(" not in q
|
|
94
|
+
assert "Self-Supervised" in q # hyphens survive
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|