bibcite-cli 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/PKG-INFO +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/pyproject.toml +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/__init__.py +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/bibfile.py +51 -4
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/cli.py +73 -28
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/normalize.py +18 -4
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/resolve.py +33 -5
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/sources.py +73 -17
- bibcite_cli-0.5.0/tests/test_round3.py +86 -0
- bibcite_cli-0.5.0/tests/test_status_semantics.py +78 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/uv.lock +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/.gitignore +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/LICENSE +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/Readme.md +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/venues.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_round2.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_venues.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -133,15 +133,37 @@ def load_bib_file(path: Path) -> BibDatabase | None:
|
|
|
133
133
|
return None
|
|
134
134
|
|
|
135
135
|
|
|
136
|
-
def find_existing(
|
|
136
|
+
def find_existing(
|
|
137
|
+
db: BibDatabase,
|
|
138
|
+
title: str,
|
|
139
|
+
arxiv_id: str = "",
|
|
140
|
+
doi: str = "",
|
|
141
|
+
author: str = "",
|
|
142
|
+
) -> dict | None:
|
|
143
|
+
from .normalize import first_author_last_name, titles_similar
|
|
144
|
+
|
|
137
145
|
ref = norm_title(title)
|
|
138
146
|
for entry in db.entries:
|
|
139
147
|
if arxiv_id and entry_arxiv_id(entry) == arxiv_id:
|
|
140
148
|
return entry
|
|
141
|
-
if doi
|
|
142
|
-
|
|
149
|
+
if doi:
|
|
150
|
+
d = doi.lower()
|
|
151
|
+
# Older entries may lack a doi field but carry it in the url.
|
|
152
|
+
if entry.get("doi", "").lower() == d or d in entry.get("url", "").lower():
|
|
153
|
+
return entry
|
|
143
154
|
if ref and norm_title(entry.get("title", "")) == ref:
|
|
144
155
|
return entry
|
|
156
|
+
# Fuzzy pass: title drift (arXiv vs camera-ready) with the same first
|
|
157
|
+
# author is the same paper — catch it BEFORE writing a duplicate pair.
|
|
158
|
+
if title and author:
|
|
159
|
+
last = first_author_last_name(author)
|
|
160
|
+
for entry in db.entries:
|
|
161
|
+
if not entry.get("author"):
|
|
162
|
+
continue
|
|
163
|
+
if first_author_last_name(entry["author"]) != last:
|
|
164
|
+
continue
|
|
165
|
+
if titles_similar(title, entry.get("title", "")):
|
|
166
|
+
return entry
|
|
145
167
|
return None
|
|
146
168
|
|
|
147
169
|
|
|
@@ -170,7 +192,11 @@ def upsert_entry(
|
|
|
170
192
|
existing = next((e for e in db.entries if e.get("ID") == replace_key), None)
|
|
171
193
|
else:
|
|
172
194
|
existing = find_existing(
|
|
173
|
-
db,
|
|
195
|
+
db,
|
|
196
|
+
entry.get("title", ""),
|
|
197
|
+
entry_arxiv_id(entry),
|
|
198
|
+
entry.get("doi", ""),
|
|
199
|
+
entry.get("author", ""),
|
|
174
200
|
)
|
|
175
201
|
|
|
176
202
|
if existing is not None:
|
|
@@ -232,7 +258,28 @@ def tidy_command() -> list[str] | None:
|
|
|
232
258
|
return None
|
|
233
259
|
|
|
234
260
|
|
|
261
|
+
_MONTH_STRING_BLOCK = re.compile(
|
|
262
|
+
r"@string\s*\{\s*(?:" + "|".join(MONTH_STRINGS) + r")\s*=",
|
|
263
|
+
re.IGNORECASE,
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _scrub_month_strings(path: Path):
|
|
268
|
+
"""Remove orphan month @string blocks left by the pre-0.4 leak.
|
|
269
|
+
bibtex-tidy itself preserves @strings, so tidy alone never cleans them."""
|
|
270
|
+
try:
|
|
271
|
+
if not _MONTH_STRING_BLOCK.search(path.read_text()):
|
|
272
|
+
return
|
|
273
|
+
db = load_bib_file(path)
|
|
274
|
+
if db is not None:
|
|
275
|
+
_write_db(path, db) # _write_db drops the injected month macros
|
|
276
|
+
_log("[bibcite] scrubbed leftover month @string blocks")
|
|
277
|
+
except Exception as e:
|
|
278
|
+
_log(f"[bibcite] month-string scrub skipped: {e}")
|
|
279
|
+
|
|
280
|
+
|
|
235
281
|
def run_tidy(path: Path) -> bool:
|
|
282
|
+
_scrub_month_strings(path)
|
|
236
283
|
cmd = tidy_command()
|
|
237
284
|
if cmd is None:
|
|
238
285
|
_log("[bibcite] bibtex-tidy not found (npm i -g bibtex-tidy); skipping tidy")
|
|
@@ -12,7 +12,7 @@ import time
|
|
|
12
12
|
from pathlib import Path
|
|
13
13
|
|
|
14
14
|
from . import bibfile, cache
|
|
15
|
-
from .normalize import first_author_last_name, norm_title, titles_similar
|
|
15
|
+
from .normalize import first_author_last_name, fix_pages, norm_title, titles_similar
|
|
16
16
|
from .resolve import classify
|
|
17
17
|
from .resolve import (
|
|
18
18
|
NotFound,
|
|
@@ -80,18 +80,18 @@ def cmd_get(args) -> int:
|
|
|
80
80
|
res, code = _resolve_or_none(query, args.require_published)
|
|
81
81
|
if res is None:
|
|
82
82
|
return code
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
)
|
|
83
|
+
payload = {
|
|
84
|
+
"action": "resolved",
|
|
85
|
+
"key": res.entry["ID"],
|
|
86
|
+
"title": res.entry.get("title", ""),
|
|
87
|
+
"venue": res.venue or "arXiv (preprint, no published venue found)",
|
|
88
|
+
"published": res.published,
|
|
89
|
+
"source": res.source,
|
|
90
|
+
"bibtex": res.bibtex,
|
|
91
|
+
}
|
|
92
|
+
if not res.published:
|
|
93
|
+
payload["published_check"] = res.check
|
|
94
|
+
_emit(payload, args.json)
|
|
95
95
|
return 0
|
|
96
96
|
|
|
97
97
|
|
|
@@ -108,6 +108,8 @@ def _resolve_user_bibtex(text: str) -> Resolved:
|
|
|
108
108
|
entry.pop("journal", None)
|
|
109
109
|
entry["ENTRYTYPE"] = canonical.entry_type
|
|
110
110
|
entry[canonical.bib_field] = canonical.name
|
|
111
|
+
if entry.get("pages"):
|
|
112
|
+
entry["pages"] = fix_pages(entry["pages"])
|
|
111
113
|
published = not bibfile.is_preprint(entry)
|
|
112
114
|
return Resolved(
|
|
113
115
|
entry,
|
|
@@ -117,6 +119,31 @@ def _resolve_user_bibtex(text: str) -> Resolved:
|
|
|
117
119
|
)
|
|
118
120
|
|
|
119
121
|
|
|
122
|
+
def _identity_mismatch(path: Path, target_key: str, new_entry: dict) -> str:
|
|
123
|
+
"""`--key` is a scalpel — warn when the resolved paper does not look like
|
|
124
|
+
the entry it is about to overwrite (no shared arXiv id, DOI, or similar
|
|
125
|
+
title), because the old key would then be citing a different paper."""
|
|
126
|
+
db = bibfile.load_bib_file(path)
|
|
127
|
+
if db is None:
|
|
128
|
+
return ""
|
|
129
|
+
target = next((e for e in db.entries if e.get("ID") == target_key), None)
|
|
130
|
+
if target is None:
|
|
131
|
+
return ""
|
|
132
|
+
old_aid, new_aid = bibfile.entry_arxiv_id(target), bibfile.entry_arxiv_id(new_entry)
|
|
133
|
+
if old_aid and new_aid and old_aid == new_aid:
|
|
134
|
+
return ""
|
|
135
|
+
old_doi, new_doi = target.get("doi", "").lower(), new_entry.get("doi", "").lower()
|
|
136
|
+
if old_doi and new_doi and old_doi == new_doi:
|
|
137
|
+
return ""
|
|
138
|
+
if titles_similar(target.get("title", ""), new_entry.get("title", "")):
|
|
139
|
+
return ""
|
|
140
|
+
return (
|
|
141
|
+
f"replacing '{target_key}' with what looks like a DIFFERENT paper "
|
|
142
|
+
f"('{target.get('title', '')[:50]}' -> '{new_entry.get('title', '')[:50]}'); "
|
|
143
|
+
"the key will no longer describe its contents"
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
120
147
|
def _local_exists(path: Path, query: str) -> str | None:
|
|
121
148
|
"""Local pre-check: if the query is already in the file as a PUBLISHED
|
|
122
149
|
entry, skip the network entirely (makes --from re-runs and repeated adds
|
|
@@ -131,7 +158,12 @@ def _local_exists(path: Path, query: str) -> str | None:
|
|
|
131
158
|
existing = bibfile.find_existing(db, "", doi=value)
|
|
132
159
|
else:
|
|
133
160
|
existing = bibfile.find_existing(db, value)
|
|
134
|
-
if existing is
|
|
161
|
+
if existing is None:
|
|
162
|
+
return None
|
|
163
|
+
if not bibfile.is_preprint(existing):
|
|
164
|
+
return existing["ID"]
|
|
165
|
+
if existing.get("pubstate", "").strip("{}") == "preprint":
|
|
166
|
+
# Confirmed preprint-only: nothing to upgrade, no reason to go online.
|
|
135
167
|
return existing["ID"]
|
|
136
168
|
return None
|
|
137
169
|
|
|
@@ -196,6 +228,11 @@ def cmd_add(args) -> int:
|
|
|
196
228
|
if res is None:
|
|
197
229
|
results.append({"query": query, "action": "failed", "exit_code": code})
|
|
198
230
|
continue
|
|
231
|
+
warning = ""
|
|
232
|
+
if args.key:
|
|
233
|
+
warning = _identity_mismatch(path, args.key, res.entry)
|
|
234
|
+
if warning:
|
|
235
|
+
_log(f"[bibcite] warning: {warning}")
|
|
199
236
|
action, key = bibfile.upsert_entry(
|
|
200
237
|
path, res.entry, replace=args.replace, replace_key=args.key or ""
|
|
201
238
|
)
|
|
@@ -210,17 +247,22 @@ def cmd_add(args) -> int:
|
|
|
210
247
|
results.append({"query": query, "action": action, "exit_code": EXIT_NOT_FOUND})
|
|
211
248
|
continue
|
|
212
249
|
wrote = wrote or action != "exists"
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
250
|
+
result = {
|
|
251
|
+
"query": query,
|
|
252
|
+
"action": action,
|
|
253
|
+
"key": key,
|
|
254
|
+
"title": res.entry.get("title", ""),
|
|
255
|
+
"venue": res.venue or "arXiv (preprint)",
|
|
256
|
+
"published": res.published,
|
|
257
|
+
"source": res.source,
|
|
258
|
+
}
|
|
259
|
+
if not res.published:
|
|
260
|
+
# "incomplete" = the check ran while core sources were down; the
|
|
261
|
+
# paper may well be published — retry later.
|
|
262
|
+
result["published_check"] = res.check
|
|
263
|
+
if warning:
|
|
264
|
+
result["warning"] = warning
|
|
265
|
+
results.append(result)
|
|
224
266
|
|
|
225
267
|
tidied = False
|
|
226
268
|
if wrote and not args.no_tidy:
|
|
@@ -274,10 +316,13 @@ def _upgrade_entries(path: Path, dry_run: bool) -> dict:
|
|
|
274
316
|
)
|
|
275
317
|
match, status = find_published(title, entry.get("year", ""), aid, hint)
|
|
276
318
|
if not match:
|
|
277
|
-
# "no_published_version" is
|
|
278
|
-
#
|
|
319
|
+
# "no_published_version" is trustworthy ONLY when every core
|
|
320
|
+
# source answered; a batch that tripped DBLP's rate limit gets
|
|
321
|
+
# "sources_unavailable" so nobody believes a poisoned verdict.
|
|
279
322
|
reason = (
|
|
280
|
-
"
|
|
323
|
+
"no_published_version"
|
|
324
|
+
if status == "not_found"
|
|
325
|
+
else "sources_unavailable"
|
|
281
326
|
)
|
|
282
327
|
report.append(
|
|
283
328
|
{"key": entry["ID"], "title": title, "matched": False, "reason": reason}
|
|
@@ -82,14 +82,22 @@ def sig_tokens(title: str) -> set[str]:
|
|
|
82
82
|
return {t for t in tokens if len(t) > 2 and t not in ENGLISH_STOPWORDS}
|
|
83
83
|
|
|
84
84
|
|
|
85
|
-
def titles_similar(a: str, b: str, threshold: float = 0.
|
|
86
|
-
"""Token-
|
|
85
|
+
def titles_similar(a: str, b: str, threshold: float = 0.75) -> bool:
|
|
86
|
+
"""Token-overlap similarity — catches preprint→camera-ready title drift
|
|
87
87
|
("Information-Theoretic Perspective" vs "Information Theory Perspective")
|
|
88
|
-
without matching genuinely different papers.
|
|
88
|
+
without matching genuinely different papers.
|
|
89
|
+
|
|
90
|
+
Uses the overlap coefficient (|∩| / min) rather than Jaccard so one
|
|
91
|
+
changed word in a shortish title still matches; very short titles
|
|
92
|
+
(<=3 significant tokens, e.g. "Deep Learning") must match exactly
|
|
93
|
+
because a single shared word would otherwise dominate."""
|
|
89
94
|
ta, tb = sig_tokens(a), sig_tokens(b)
|
|
90
95
|
if not ta or not tb:
|
|
91
96
|
return False
|
|
92
|
-
|
|
97
|
+
smaller = min(len(ta), len(tb))
|
|
98
|
+
if smaller <= 3:
|
|
99
|
+
return ta == tb
|
|
100
|
+
return len(ta & tb) / smaller >= threshold
|
|
93
101
|
|
|
94
102
|
|
|
95
103
|
def fix_author_caps(author_field: str) -> str:
|
|
@@ -114,6 +122,12 @@ def fix_author_caps(author_field: str) -> str:
|
|
|
114
122
|
return " and ".join(fix_name(n) for n in names)
|
|
115
123
|
|
|
116
124
|
|
|
125
|
+
def fix_pages(pages: str) -> str:
|
|
126
|
+
"""BibTeX page ranges use `--`; CrossRef emits en-dashes (411–430) and
|
|
127
|
+
some sources a single hyphen. Collapse any dash run to `--`."""
|
|
128
|
+
return re.sub(r"\s*[-‐-―]+\s*", "--", pages.strip())
|
|
129
|
+
|
|
130
|
+
|
|
117
131
|
def make_key(author_field: str, year: str | int, title: str) -> str:
|
|
118
132
|
"""Deterministic citation key: <lastname><year><firstword>.
|
|
119
133
|
|
|
@@ -10,7 +10,13 @@ import sys
|
|
|
10
10
|
from dataclasses import dataclass
|
|
11
11
|
|
|
12
12
|
from .bibfile import NOISE_FIELDS, parse_bibtex_entry
|
|
13
|
-
from .normalize import
|
|
13
|
+
from .normalize import (
|
|
14
|
+
clean_title,
|
|
15
|
+
first_author_last_name,
|
|
16
|
+
fix_author_caps,
|
|
17
|
+
fix_pages,
|
|
18
|
+
make_key,
|
|
19
|
+
)
|
|
14
20
|
|
|
15
21
|
|
|
16
22
|
class NotFound(Exception):
|
|
@@ -66,6 +72,12 @@ class Resolved:
|
|
|
66
72
|
source: str # where the publication info came from
|
|
67
73
|
venue: str # final venue string ("" if preprint)
|
|
68
74
|
published: bool
|
|
75
|
+
# For preprint fallbacks: was the publication check trustworthy?
|
|
76
|
+
# "complete" — every core source answered; the paper really has no
|
|
77
|
+
# published version (as of now)
|
|
78
|
+
# "incomplete" — core sources were rate-limited/down; retry later before
|
|
79
|
+
# believing the preprint status
|
|
80
|
+
check: str = "complete"
|
|
69
81
|
|
|
70
82
|
@property
|
|
71
83
|
def bibtex(self) -> str:
|
|
@@ -153,6 +165,8 @@ def _finalize(entry: dict, meta: ArxivMeta | None) -> dict:
|
|
|
153
165
|
# a missing url from the DOI.
|
|
154
166
|
if not url or "dx.doi.org" in url:
|
|
155
167
|
entry["url"] = f"https://doi.org/{entry['doi']}"
|
|
168
|
+
if entry.get("pages"):
|
|
169
|
+
entry["pages"] = fix_pages(entry["pages"])
|
|
156
170
|
author = entry.get("author", "") or "anonymous"
|
|
157
171
|
year = entry.get("year", "") or "XXXX"
|
|
158
172
|
entry["ID"] = make_key(author, year, entry.get("title", ""))
|
|
@@ -212,9 +226,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
|
|
|
212
226
|
f"Could not check publication status for arXiv:{value} (sources down)"
|
|
213
227
|
)
|
|
214
228
|
raise NotFound(f"No published version found for arXiv:{value}")
|
|
215
|
-
|
|
229
|
+
check = "complete" if status == "not_found" else "incomplete"
|
|
230
|
+
if check == "incomplete":
|
|
231
|
+
_log(
|
|
232
|
+
"[bibcite] preprint fallback with INCOMPLETE publication check "
|
|
233
|
+
"(core sources were unavailable) — retry later"
|
|
234
|
+
)
|
|
235
|
+
else:
|
|
236
|
+
_log("[bibcite] no published version found; using arXiv preprint entry")
|
|
216
237
|
entry = _arxiv_only_entry(meta)
|
|
217
|
-
return Resolved(_finalize(entry, meta), "arxiv", "", False)
|
|
238
|
+
return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
|
|
218
239
|
|
|
219
240
|
if kind == "doi":
|
|
220
241
|
match = crossref_by_doi(value)
|
|
@@ -248,9 +269,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
|
|
|
248
269
|
if meta and meta.arxiv_id:
|
|
249
270
|
if require_published:
|
|
250
271
|
raise NotFound(f"Only an arXiv preprint was found for: {value}")
|
|
251
|
-
|
|
272
|
+
check = "complete" if status == "not_found" else "incomplete"
|
|
273
|
+
if check == "incomplete":
|
|
274
|
+
_log(
|
|
275
|
+
"[bibcite] preprint fallback with INCOMPLETE publication check "
|
|
276
|
+
"(core sources were unavailable) — retry later"
|
|
277
|
+
)
|
|
278
|
+
else:
|
|
279
|
+
_log("[bibcite] no published version found; using arXiv preprint entry")
|
|
252
280
|
entry = _arxiv_only_entry(meta)
|
|
253
|
-
return Resolved(_finalize(entry, meta), "arxiv", "", False)
|
|
281
|
+
return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
|
|
254
282
|
if status == "unavailable":
|
|
255
283
|
raise SourcesUnavailable(
|
|
256
284
|
f"All sources were rate-limited or down while resolving: {value}"
|
|
@@ -147,6 +147,42 @@ def arxiv_metadata(arxiv_id: str) -> ArxivMeta:
|
|
|
147
147
|
# DBLP
|
|
148
148
|
# ---------------------------------------------------------------------------
|
|
149
149
|
|
|
150
|
+
# DBLP throttles at roughly 1-2 req/s and escalates to temporary IP bans when
|
|
151
|
+
# hammered. Client-side pacing prevents the 429 in the first place; on a 429
|
|
152
|
+
# we back off and retry (honoring Retry-After) instead of instantly poisoning
|
|
153
|
+
# the rest of a batch run — only repeated failure disables the source.
|
|
154
|
+
_DBLP_MIN_INTERVAL = 0.8
|
|
155
|
+
_dblp_last_request = 0.0
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _dblp_get(c: httpx.Client, url: str, params: dict | None = None) -> httpx.Response:
|
|
159
|
+
global _dblp_last_request
|
|
160
|
+
for attempt in range(3):
|
|
161
|
+
wait = _DBLP_MIN_INTERVAL - (time.monotonic() - _dblp_last_request)
|
|
162
|
+
if wait > 0:
|
|
163
|
+
time.sleep(wait)
|
|
164
|
+
_dblp_last_request = time.monotonic()
|
|
165
|
+
try:
|
|
166
|
+
r = c.get(url, params=params)
|
|
167
|
+
except httpx.HTTPError as e: # TCP reset = temporary ban; retrying fast makes it worse
|
|
168
|
+
if attempt < 2:
|
|
169
|
+
time.sleep(5 * (attempt + 1))
|
|
170
|
+
continue
|
|
171
|
+
raise SourceUnavailable(f"DBLP unreachable ({type(e).__name__})")
|
|
172
|
+
if r.status_code == 429:
|
|
173
|
+
retry_after = int(r.headers.get("Retry-After") or 0)
|
|
174
|
+
if retry_after > 30:
|
|
175
|
+
raise SourceUnavailable(f"DBLP rate-limited (Retry-After {retry_after}s)")
|
|
176
|
+
if attempt < 2:
|
|
177
|
+
delay = max(retry_after, 4 * (attempt + 1))
|
|
178
|
+
_log(f"[dblp] 429 — backing off {delay}s")
|
|
179
|
+
time.sleep(delay)
|
|
180
|
+
continue
|
|
181
|
+
raise SourceUnavailable("DBLP rate-limited (429) after backoff retries")
|
|
182
|
+
return r
|
|
183
|
+
raise SourceUnavailable("DBLP unavailable")
|
|
184
|
+
|
|
185
|
+
|
|
150
186
|
def try_dblp(title: str, author_hint: str = "") -> Match | None:
|
|
151
187
|
"""DBLP search. Generic titles ("X is all you need") drown in DBLP's
|
|
152
188
|
ranking, so when we know the first author we query with their last name
|
|
@@ -157,12 +193,11 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
|
|
|
157
193
|
queries.append(title)
|
|
158
194
|
with _client() as c:
|
|
159
195
|
for q in queries:
|
|
160
|
-
r =
|
|
196
|
+
r = _dblp_get(
|
|
197
|
+
c,
|
|
161
198
|
"https://dblp.org/search/publ/api",
|
|
162
199
|
params={"q": q, "format": "json", "h": 100},
|
|
163
200
|
)
|
|
164
|
-
if r.status_code == 429:
|
|
165
|
-
raise SourceUnavailable("DBLP rate-limited (429)")
|
|
166
201
|
r.raise_for_status()
|
|
167
202
|
hits = (
|
|
168
203
|
r.json().get("result", {}).get("hits", {}).get("hit", []) or []
|
|
@@ -182,9 +217,12 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
|
|
|
182
217
|
venue = venue[0]
|
|
183
218
|
bibtex = ""
|
|
184
219
|
if info.get("url"):
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
220
|
+
try:
|
|
221
|
+
br = _dblp_get(c, info["url"] + ".bib")
|
|
222
|
+
if br.status_code == 200:
|
|
223
|
+
bibtex = br.text
|
|
224
|
+
except SourceUnavailable:
|
|
225
|
+
pass # keep the match; construct from fields
|
|
188
226
|
_log(f"[dblp] match: {venue} {info.get('year', '')}")
|
|
189
227
|
return Match(
|
|
190
228
|
source="dblp",
|
|
@@ -219,12 +257,11 @@ def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None
|
|
|
219
257
|
return None
|
|
220
258
|
q = " ".join([author_hint] + tokens)
|
|
221
259
|
with _client() as c:
|
|
222
|
-
r =
|
|
260
|
+
r = _dblp_get(
|
|
261
|
+
c,
|
|
223
262
|
"https://dblp.org/search/publ/api",
|
|
224
263
|
params={"q": q, "format": "json", "h": 100},
|
|
225
264
|
)
|
|
226
|
-
if r.status_code == 429:
|
|
227
|
-
raise SourceUnavailable("DBLP rate-limited (429)")
|
|
228
265
|
r.raise_for_status()
|
|
229
266
|
hits = r.json().get("result", {}).get("hits", {}).get("hit", []) or []
|
|
230
267
|
hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
|
|
@@ -246,9 +283,12 @@ def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None
|
|
|
246
283
|
venue = venue[0]
|
|
247
284
|
bibtex = ""
|
|
248
285
|
if info.get("url"):
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
286
|
+
try:
|
|
287
|
+
br = _dblp_get(c, info["url"] + ".bib")
|
|
288
|
+
if br.status_code == 200:
|
|
289
|
+
bibtex = br.text
|
|
290
|
+
except SourceUnavailable:
|
|
291
|
+
pass # keep the match; construct from fields
|
|
252
292
|
_log(
|
|
253
293
|
f"[dblp-fuzzy] match with title drift: '{hit_title}' "
|
|
254
294
|
f"@ {venue} {info.get('year', '')}"
|
|
@@ -659,16 +699,24 @@ CASCADE = (
|
|
|
659
699
|
# (PaperMemory's DISABLE_MATCH, ported).
|
|
660
700
|
_DISABLED: dict[str, str] = {}
|
|
661
701
|
|
|
702
|
+
# Only these sources are authoritative enough that losing one taints a miss
|
|
703
|
+
# into "incomplete". Google Scholar captchas and Unpaywall flakiness are
|
|
704
|
+
# routine and must not stop "not_found" from ever being trustworthy.
|
|
705
|
+
CORE_SOURCES = frozenset({"dblp", "semanticscholar", "crossref", "openalex"})
|
|
706
|
+
|
|
662
707
|
|
|
663
708
|
def find_published(
|
|
664
709
|
title: str, year: str = "", arxiv_id: str = "", author_hint: str = ""
|
|
665
710
|
) -> tuple[Match | None, str]:
|
|
666
711
|
"""Try each source in order; first verified hit wins.
|
|
667
712
|
|
|
668
|
-
Returns (match, status)
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
713
|
+
Returns (match, status):
|
|
714
|
+
"found" — verified publication match
|
|
715
|
+
"not_found" — EVERY source answered cleanly with no hit; trustworthy
|
|
716
|
+
"incomplete" — some sources answered (no hit) but others were
|
|
717
|
+
disabled/erroring; a batch run that tripped DBLP's rate
|
|
718
|
+
limit lands here — do NOT conclude "unpublished"
|
|
719
|
+
"unavailable" — no source answered at all
|
|
672
720
|
"""
|
|
673
721
|
from . import cache
|
|
674
722
|
|
|
@@ -679,6 +727,8 @@ def find_published(
|
|
|
679
727
|
return Match(**cached), "found"
|
|
680
728
|
|
|
681
729
|
clean_misses = 0
|
|
730
|
+
# Core sources lost earlier in this run taint this query's verdict too.
|
|
731
|
+
incomplete = any(n in CORE_SOURCES for n in _DISABLED)
|
|
682
732
|
for name, fn in CASCADE:
|
|
683
733
|
if name in _DISABLED:
|
|
684
734
|
continue
|
|
@@ -691,8 +741,10 @@ def find_published(
|
|
|
691
741
|
_log(f"[{name}] no publication found")
|
|
692
742
|
except SourceUnavailable as e:
|
|
693
743
|
_DISABLED[name] = str(e)
|
|
744
|
+
incomplete = incomplete or name in CORE_SOURCES
|
|
694
745
|
_log(f"[{name}] disabled for the rest of this run: {e}")
|
|
695
746
|
except Exception as e: # network hiccup on one source must not kill the run
|
|
747
|
+
incomplete = incomplete or name in CORE_SOURCES
|
|
696
748
|
_log(f"[{name}] error: {type(e).__name__}: {e}")
|
|
697
749
|
|
|
698
750
|
# Exact-title search missed everywhere. Before concluding "no published
|
|
@@ -707,6 +759,10 @@ def find_published(
|
|
|
707
759
|
clean_misses += 1
|
|
708
760
|
except SourceUnavailable as e:
|
|
709
761
|
_DISABLED["dblp"] = str(e)
|
|
762
|
+
incomplete = True
|
|
710
763
|
except Exception as e:
|
|
764
|
+
incomplete = True
|
|
711
765
|
_log(f"[dblp-fuzzy] error: {type(e).__name__}: {e}")
|
|
712
|
-
|
|
766
|
+
if not clean_misses:
|
|
767
|
+
return None, "unavailable"
|
|
768
|
+
return None, ("incomplete" if incomplete else "not_found")
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Regression tests for the third round of field-use reports."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from bibcite.bibfile import (
|
|
6
|
+
_scrub_month_strings,
|
|
7
|
+
find_existing,
|
|
8
|
+
load_bib_file,
|
|
9
|
+
upsert_entry,
|
|
10
|
+
)
|
|
11
|
+
from bibcite.normalize import fix_pages
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_fix_pages_dashes():
|
|
15
|
+
assert fix_pages("411–430") == "411--430" # en-dash
|
|
16
|
+
assert fix_pages("411-430") == "411--430" # single hyphen
|
|
17
|
+
assert fix_pages("411 -- 430") == "411--430"
|
|
18
|
+
assert fix_pages("723—726") == "723--726" # em-dash
|
|
19
|
+
assert fix_pages("e123") == "e123" # no range untouched
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
PREPRINT = {
|
|
23
|
+
"ENTRYTYPE": "misc",
|
|
24
|
+
"ID": "old",
|
|
25
|
+
"title": "An Information-Theoretic Perspective on VICReg",
|
|
26
|
+
"author": "Ravid Shwartz-Ziv and Yann LeCun",
|
|
27
|
+
"howpublished": "arXiv preprint arXiv:2303.00633",
|
|
28
|
+
"eprint": "2303.00633",
|
|
29
|
+
"year": "2023",
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_find_existing_by_doi_in_url(tmp_path: Path):
|
|
34
|
+
bib = tmp_path / "d.bib"
|
|
35
|
+
upsert_entry(
|
|
36
|
+
bib,
|
|
37
|
+
{
|
|
38
|
+
"ENTRYTYPE": "article",
|
|
39
|
+
"ID": "k",
|
|
40
|
+
"title": "T",
|
|
41
|
+
"author": "A B",
|
|
42
|
+
"journal": "J",
|
|
43
|
+
"year": "2000",
|
|
44
|
+
"url": "https://doi.org/10.1093/biomet/70.3.723",
|
|
45
|
+
},
|
|
46
|
+
)
|
|
47
|
+
db = load_bib_file(bib)
|
|
48
|
+
# No doi field on the entry — matched via the url (pre-0.4.0 files).
|
|
49
|
+
assert find_existing(db, "", doi="10.1093/biomet/70.3.723") is not None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_dedupe_catches_title_drift_pair(tmp_path: Path):
|
|
53
|
+
bib = tmp_path / "p.bib"
|
|
54
|
+
upsert_entry(bib, dict(PREPRINT))
|
|
55
|
+
published = {
|
|
56
|
+
"ENTRYTYPE": "inproceedings",
|
|
57
|
+
"ID": "new",
|
|
58
|
+
"title": "An Information Theory Perspective on VICReg", # drifted
|
|
59
|
+
"author": "Ravid Shwartz-Ziv and Yann LeCun",
|
|
60
|
+
"booktitle": "Advances in Neural Information Processing Systems (NeurIPS)",
|
|
61
|
+
"year": "2023",
|
|
62
|
+
}
|
|
63
|
+
action, key = upsert_entry(bib, published)
|
|
64
|
+
# Fuzzy same-author dedupe: upgraded in place, NOT added as a duplicate.
|
|
65
|
+
assert (action, key) == ("upgraded", "old")
|
|
66
|
+
assert bib.read_text().count("@") == 1
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_scrub_orphan_month_strings(tmp_path: Path):
|
|
70
|
+
bib = tmp_path / "m.bib"
|
|
71
|
+
bib.write_text(
|
|
72
|
+
"@string{january = {January}}\n@string{june = {June}}\n"
|
|
73
|
+
"@article{x, title = {T}, author = {A B}, year = {2000} }\n"
|
|
74
|
+
)
|
|
75
|
+
_scrub_month_strings(bib)
|
|
76
|
+
text = bib.read_text()
|
|
77
|
+
assert "@string" not in text
|
|
78
|
+
assert "title" in text
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_scrub_leaves_clean_files_alone(tmp_path: Path):
|
|
82
|
+
bib = tmp_path / "c.bib"
|
|
83
|
+
original = "@article{x,\n title = {T},\n author = {A B},\n year = {2000},\n}\n"
|
|
84
|
+
bib.write_text(original)
|
|
85
|
+
_scrub_month_strings(bib)
|
|
86
|
+
assert bib.read_text() == original # untouched, not even rewritten
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""The batch-429 poisoning scenario: a disabled core source must taint the
|
|
2
|
+
verdict ('incomplete'), never masquerade as a trustworthy 'not_found'."""
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
import bibcite.sources as sources
|
|
7
|
+
from bibcite import cache
|
|
8
|
+
from bibcite.sources import SourceUnavailable, find_published
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@pytest.fixture(autouse=True)
|
|
12
|
+
def isolated(monkeypatch, tmp_path):
|
|
13
|
+
monkeypatch.setattr(cache, "DISABLED", True)
|
|
14
|
+
monkeypatch.setattr(sources, "_DISABLED", {})
|
|
15
|
+
# No fuzzy fallback network calls in these tests.
|
|
16
|
+
monkeypatch.setattr(sources, "try_dblp_fuzzy", lambda *a, **k: None)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _cascade(**outcomes):
|
|
20
|
+
"""Build a fake CASCADE; outcome per source: None=clean miss, 'raise'=429."""
|
|
21
|
+
|
|
22
|
+
def make(o):
|
|
23
|
+
def fn(t, y, a, au):
|
|
24
|
+
if o == "raise":
|
|
25
|
+
raise SourceUnavailable("simulated 429")
|
|
26
|
+
return o
|
|
27
|
+
|
|
28
|
+
return fn
|
|
29
|
+
|
|
30
|
+
return tuple((name, make(o)) for name, o in outcomes.items())
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_all_clean_misses_is_trustworthy(monkeypatch):
|
|
34
|
+
monkeypatch.setattr(
|
|
35
|
+
sources, "CASCADE", _cascade(dblp=None, semanticscholar=None, crossref=None)
|
|
36
|
+
)
|
|
37
|
+
match, status = find_published("Some Title", author_hint="smith")
|
|
38
|
+
assert (match, status) == (None, "not_found")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_core_source_429_taints_verdict(monkeypatch):
|
|
42
|
+
# DBLP 429s, others answer cleanly — the user's exact failure chain.
|
|
43
|
+
monkeypatch.setattr(
|
|
44
|
+
sources,
|
|
45
|
+
"CASCADE",
|
|
46
|
+
_cascade(dblp="raise", googlescholar=None, crossref=None),
|
|
47
|
+
)
|
|
48
|
+
match, status = find_published("Some Title", author_hint="smith")
|
|
49
|
+
assert (match, status) == (None, "incomplete")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_previously_disabled_core_source_taints_next_queries(monkeypatch):
|
|
53
|
+
# Query N tripped DBLP; queries N+1... in the same run inherit the taint.
|
|
54
|
+
monkeypatch.setattr(sources, "_DISABLED", {"dblp": "429"})
|
|
55
|
+
monkeypatch.setattr(
|
|
56
|
+
sources, "CASCADE", _cascade(dblp=None, crossref=None) # dblp skipped anyway
|
|
57
|
+
)
|
|
58
|
+
match, status = find_published("Another Title", author_hint="smith")
|
|
59
|
+
assert (match, status) == (None, "incomplete")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_noncore_outage_does_not_taint(monkeypatch):
|
|
63
|
+
# Google Scholar captcha is routine; a miss stays trustworthy.
|
|
64
|
+
monkeypatch.setattr(
|
|
65
|
+
sources,
|
|
66
|
+
"CASCADE",
|
|
67
|
+
_cascade(dblp=None, googlescholar="raise", crossref=None),
|
|
68
|
+
)
|
|
69
|
+
match, status = find_published("Some Title", author_hint="smith")
|
|
70
|
+
assert (match, status) == (None, "not_found")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def test_everything_down_is_unavailable(monkeypatch):
|
|
74
|
+
monkeypatch.setattr(
|
|
75
|
+
sources, "CASCADE", _cascade(dblp="raise", semanticscholar="raise")
|
|
76
|
+
)
|
|
77
|
+
match, status = find_published("Some Title", author_hint="smith")
|
|
78
|
+
assert (match, status) == (None, "unavailable")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|