bibcite-cli 0.3.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/PKG-INFO +1 -1
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/pyproject.toml +1 -1
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/__init__.py +1 -1
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/bibfile.py +83 -13
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/cli.py +141 -21
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/normalize.py +30 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/resolve.py +9 -1
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/sources.py +83 -1
- bibcite_cli-0.4.1/tests/test_round2.py +70 -0
- bibcite_cli-0.4.1/tests/test_round3.py +86 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/uv.lock +1 -1
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/.gitignore +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/LICENSE +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/Readme.md +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/venues.py +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_venues.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -20,14 +20,16 @@ from .normalize import norm_title
|
|
|
20
20
|
# \cite{} commands valid.
|
|
21
21
|
TIDY_ARGS = [
|
|
22
22
|
"--modify",
|
|
23
|
-
|
|
23
|
+
# volume/number/pages/doi are kept (bibliographic substance the user
|
|
24
|
+
# asked to retain); the omit list drops only true noise.
|
|
25
|
+
"--omit=publisher,timestamp,biburl,bibsource,abstract,month,series,editor,note,date,address",
|
|
24
26
|
"--curly",
|
|
25
27
|
"--blank-lines",
|
|
26
28
|
"--trailing-commas",
|
|
27
29
|
"--sort=-year",
|
|
28
30
|
"--duplicates=citation",
|
|
29
31
|
"--merge=first",
|
|
30
|
-
"--sort-fields=author,title,booktitle,journal,year,url,pdf",
|
|
32
|
+
"--sort-fields=author,title,booktitle,journal,volume,number,pages,year,doi,url,pdf",
|
|
31
33
|
"--strip-enclosing-braces",
|
|
32
34
|
"--tidy-comments",
|
|
33
35
|
]
|
|
@@ -131,45 +133,86 @@ def load_bib_file(path: Path) -> BibDatabase | None:
|
|
|
131
133
|
return None
|
|
132
134
|
|
|
133
135
|
|
|
134
|
-
def find_existing(
|
|
136
|
+
def find_existing(
|
|
137
|
+
db: BibDatabase,
|
|
138
|
+
title: str,
|
|
139
|
+
arxiv_id: str = "",
|
|
140
|
+
doi: str = "",
|
|
141
|
+
author: str = "",
|
|
142
|
+
) -> dict | None:
|
|
143
|
+
from .normalize import first_author_last_name, titles_similar
|
|
144
|
+
|
|
135
145
|
ref = norm_title(title)
|
|
136
146
|
for entry in db.entries:
|
|
137
147
|
if arxiv_id and entry_arxiv_id(entry) == arxiv_id:
|
|
138
148
|
return entry
|
|
139
|
-
if doi
|
|
140
|
-
|
|
149
|
+
if doi:
|
|
150
|
+
d = doi.lower()
|
|
151
|
+
# Older entries may lack a doi field but carry it in the url.
|
|
152
|
+
if entry.get("doi", "").lower() == d or d in entry.get("url", "").lower():
|
|
153
|
+
return entry
|
|
141
154
|
if ref and norm_title(entry.get("title", "")) == ref:
|
|
142
155
|
return entry
|
|
156
|
+
# Fuzzy pass: title drift (arXiv vs camera-ready) with the same first
|
|
157
|
+
# author is the same paper — catch it BEFORE writing a duplicate pair.
|
|
158
|
+
if title and author:
|
|
159
|
+
last = first_author_last_name(author)
|
|
160
|
+
for entry in db.entries:
|
|
161
|
+
if not entry.get("author"):
|
|
162
|
+
continue
|
|
163
|
+
if first_author_last_name(entry["author"]) != last:
|
|
164
|
+
continue
|
|
165
|
+
if titles_similar(title, entry.get("title", "")):
|
|
166
|
+
return entry
|
|
143
167
|
return None
|
|
144
168
|
|
|
145
169
|
|
|
146
|
-
def upsert_entry(
|
|
170
|
+
def upsert_entry(
|
|
171
|
+
path: Path, entry: dict, replace: bool = False, replace_key: str = ""
|
|
172
|
+
) -> tuple[str, str]:
|
|
147
173
|
"""Insert or upgrade ``entry`` in ``path``.
|
|
148
174
|
|
|
149
175
|
Returns (action, key), action in "added" | "upgraded" | "exists" |
|
|
150
|
-
"replaced". With ``replace``, an existing
|
|
151
|
-
|
|
176
|
+
"replaced" | "no_match_to_replace". With ``replace``, an existing
|
|
177
|
+
matching entry is overwritten; ``replace_key`` targets a specific entry
|
|
178
|
+
by citation key (for when title drift defeats the automatic match). The
|
|
179
|
+
existing key is always kept so \\cite{} commands stay valid. A replace
|
|
180
|
+
that matches nothing is an ERROR, not a silent add — that is how
|
|
181
|
+
duplicate entries sneak into a file.
|
|
152
182
|
"""
|
|
153
183
|
db = load_bib_file(path)
|
|
154
184
|
if db is None: # unparseable file: append blindly
|
|
185
|
+
if replace or replace_key:
|
|
186
|
+
return "no_match_to_replace", replace_key or entry["ID"]
|
|
155
187
|
with path.open("a") as f:
|
|
156
188
|
f.write("\n" + entry_to_bibtex(entry))
|
|
157
189
|
return "added", entry["ID"]
|
|
158
190
|
|
|
159
|
-
|
|
160
|
-
db
|
|
161
|
-
|
|
191
|
+
if replace_key:
|
|
192
|
+
existing = next((e for e in db.entries if e.get("ID") == replace_key), None)
|
|
193
|
+
else:
|
|
194
|
+
existing = find_existing(
|
|
195
|
+
db,
|
|
196
|
+
entry.get("title", ""),
|
|
197
|
+
entry_arxiv_id(entry),
|
|
198
|
+
entry.get("doi", ""),
|
|
199
|
+
entry.get("author", ""),
|
|
200
|
+
)
|
|
201
|
+
|
|
162
202
|
if existing is not None:
|
|
163
203
|
upgrade = is_preprint(existing) and not is_preprint(entry)
|
|
164
|
-
if replace or upgrade:
|
|
204
|
+
if replace or replace_key or upgrade:
|
|
165
205
|
key = existing["ID"]
|
|
166
206
|
existing.clear()
|
|
167
207
|
existing.update({k: str(v) for k, v in entry.items() if v})
|
|
168
208
|
existing["ID"] = key # keep the key the user may already \cite
|
|
169
209
|
_write_db(path, db)
|
|
170
|
-
return ("replaced" if replace else "upgraded"), key
|
|
210
|
+
return ("replaced" if (replace or replace_key) else "upgraded"), key
|
|
171
211
|
return "exists", existing["ID"]
|
|
172
212
|
|
|
213
|
+
if replace or replace_key:
|
|
214
|
+
return "no_match_to_replace", replace_key or entry["ID"]
|
|
215
|
+
|
|
173
216
|
db.entries.append({k: str(v) for k, v in entry.items() if v})
|
|
174
217
|
_write_db(path, db)
|
|
175
218
|
return "added", entry["ID"]
|
|
@@ -190,6 +233,12 @@ def remove_entry(path: Path, key: str) -> bool:
|
|
|
190
233
|
|
|
191
234
|
|
|
192
235
|
def _write_db(path: Path, db: BibDatabase):
|
|
236
|
+
# Never write our injected month macros back out as @string blocks (they
|
|
237
|
+
# exist only so parsing month=June doesn't crash); this also scrubs any
|
|
238
|
+
# that leaked into a file before this guard existed. User-defined
|
|
239
|
+
# @strings are untouched.
|
|
240
|
+
for k in MONTH_STRINGS:
|
|
241
|
+
db.strings.pop(k, None)
|
|
193
242
|
writer = BibTexWriter()
|
|
194
243
|
writer.indent = " "
|
|
195
244
|
writer.order_entries_by = None # preserve file order; tidy re-sorts anyway
|
|
@@ -209,7 +258,28 @@ def tidy_command() -> list[str] | None:
|
|
|
209
258
|
return None
|
|
210
259
|
|
|
211
260
|
|
|
261
|
+
_MONTH_STRING_BLOCK = re.compile(
|
|
262
|
+
r"@string\s*\{\s*(?:" + "|".join(MONTH_STRINGS) + r")\s*=",
|
|
263
|
+
re.IGNORECASE,
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _scrub_month_strings(path: Path):
|
|
268
|
+
"""Remove orphan month @string blocks left by the pre-0.4 leak.
|
|
269
|
+
bibtex-tidy itself preserves @strings, so tidy alone never cleans them."""
|
|
270
|
+
try:
|
|
271
|
+
if not _MONTH_STRING_BLOCK.search(path.read_text()):
|
|
272
|
+
return
|
|
273
|
+
db = load_bib_file(path)
|
|
274
|
+
if db is not None:
|
|
275
|
+
_write_db(path, db) # _write_db drops the injected month macros
|
|
276
|
+
_log("[bibcite] scrubbed leftover month @string blocks")
|
|
277
|
+
except Exception as e:
|
|
278
|
+
_log(f"[bibcite] month-string scrub skipped: {e}")
|
|
279
|
+
|
|
280
|
+
|
|
212
281
|
def run_tidy(path: Path) -> bool:
|
|
282
|
+
_scrub_month_strings(path)
|
|
213
283
|
cmd = tidy_command()
|
|
214
284
|
if cmd is None:
|
|
215
285
|
_log("[bibcite] bibtex-tidy not found (npm i -g bibtex-tidy); skipping tidy")
|
|
@@ -12,7 +12,8 @@ import time
|
|
|
12
12
|
from pathlib import Path
|
|
13
13
|
|
|
14
14
|
from . import bibfile, cache
|
|
15
|
-
from .normalize import first_author_last_name, norm_title
|
|
15
|
+
from .normalize import first_author_last_name, fix_pages, norm_title, titles_similar
|
|
16
|
+
from .resolve import classify
|
|
16
17
|
from .resolve import (
|
|
17
18
|
NotFound,
|
|
18
19
|
Resolved,
|
|
@@ -107,62 +108,157 @@ def _resolve_user_bibtex(text: str) -> Resolved:
|
|
|
107
108
|
entry.pop("journal", None)
|
|
108
109
|
entry["ENTRYTYPE"] = canonical.entry_type
|
|
109
110
|
entry[canonical.bib_field] = canonical.name
|
|
110
|
-
|
|
111
|
+
if entry.get("pages"):
|
|
112
|
+
entry["pages"] = fix_pages(entry["pages"])
|
|
113
|
+
published = not bibfile.is_preprint(entry)
|
|
114
|
+
return Resolved(
|
|
115
|
+
entry,
|
|
116
|
+
"user-bibtex",
|
|
117
|
+
(canonical.name if canonical else raw_venue) if published else "",
|
|
118
|
+
published,
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _identity_mismatch(path: Path, target_key: str, new_entry: dict) -> str:
|
|
123
|
+
"""`--key` is a scalpel — warn when the resolved paper does not look like
|
|
124
|
+
the entry it is about to overwrite (no shared arXiv id, DOI, or similar
|
|
125
|
+
title), because the old key would then be citing a different paper."""
|
|
126
|
+
db = bibfile.load_bib_file(path)
|
|
127
|
+
if db is None:
|
|
128
|
+
return ""
|
|
129
|
+
target = next((e for e in db.entries if e.get("ID") == target_key), None)
|
|
130
|
+
if target is None:
|
|
131
|
+
return ""
|
|
132
|
+
old_aid, new_aid = bibfile.entry_arxiv_id(target), bibfile.entry_arxiv_id(new_entry)
|
|
133
|
+
if old_aid and new_aid and old_aid == new_aid:
|
|
134
|
+
return ""
|
|
135
|
+
old_doi, new_doi = target.get("doi", "").lower(), new_entry.get("doi", "").lower()
|
|
136
|
+
if old_doi and new_doi and old_doi == new_doi:
|
|
137
|
+
return ""
|
|
138
|
+
if titles_similar(target.get("title", ""), new_entry.get("title", "")):
|
|
139
|
+
return ""
|
|
140
|
+
return (
|
|
141
|
+
f"replacing '{target_key}' with what looks like a DIFFERENT paper "
|
|
142
|
+
f"('{target.get('title', '')[:50]}' -> '{new_entry.get('title', '')[:50]}'); "
|
|
143
|
+
"the key will no longer describe its contents"
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _local_exists(path: Path, query: str) -> str | None:
|
|
148
|
+
"""Local pre-check: if the query is already in the file as a PUBLISHED
|
|
149
|
+
entry, skip the network entirely (makes --from re-runs and repeated adds
|
|
150
|
+
near-instant). Preprints still resolve online — they may be upgradable."""
|
|
151
|
+
db = bibfile.load_bib_file(path)
|
|
152
|
+
if db is None or not db.entries:
|
|
153
|
+
return None
|
|
154
|
+
kind, value = classify(query)
|
|
155
|
+
if kind == "arxiv":
|
|
156
|
+
existing = bibfile.find_existing(db, "", arxiv_id=value)
|
|
157
|
+
elif kind == "doi":
|
|
158
|
+
existing = bibfile.find_existing(db, "", doi=value)
|
|
159
|
+
else:
|
|
160
|
+
existing = bibfile.find_existing(db, value)
|
|
161
|
+
if existing is None:
|
|
162
|
+
return None
|
|
163
|
+
if not bibfile.is_preprint(existing):
|
|
164
|
+
return existing["ID"]
|
|
165
|
+
if existing.get("pubstate", "").strip("{}") == "preprint":
|
|
166
|
+
# Confirmed preprint-only: nothing to upgrade, no reason to go online.
|
|
167
|
+
return existing["ID"]
|
|
168
|
+
return None
|
|
111
169
|
|
|
112
170
|
|
|
113
171
|
def cmd_add(args) -> int:
|
|
114
172
|
path = Path(args.file)
|
|
115
173
|
if args.no_cache:
|
|
116
174
|
cache.DISABLED = True
|
|
175
|
+
targeting = args.replace or bool(args.key)
|
|
176
|
+
if args.key and args.from_file:
|
|
177
|
+
_log("[bibcite] --key targets one entry; it cannot be combined with --from")
|
|
178
|
+
return EXIT_NOT_FOUND
|
|
117
179
|
|
|
118
180
|
# Collect the queries for this invocation (single, --bibtex, or --from).
|
|
181
|
+
# Each item: (query, resolved_or_None, exit_code, local_exists_key).
|
|
182
|
+
items: list[tuple[str, Resolved | None, int, str]] = []
|
|
119
183
|
if args.bibtex:
|
|
120
184
|
text = sys.stdin.read() if args.bibtex == "-" else args.bibtex
|
|
121
185
|
try:
|
|
122
|
-
|
|
186
|
+
items.append(("<bibtex>", _resolve_user_bibtex(text), 0, ""))
|
|
123
187
|
except ValueError as e:
|
|
124
188
|
_log(f"[bibcite] {e}")
|
|
125
189
|
return EXIT_NOT_FOUND
|
|
126
190
|
elif args.from_file:
|
|
127
191
|
lines = Path(args.from_file).read_text().splitlines()
|
|
128
192
|
queries = [q.strip() for q in lines if q.strip() and not q.strip().startswith("#")]
|
|
129
|
-
|
|
193
|
+
resolved_any = False
|
|
130
194
|
for i, q in enumerate(queries):
|
|
131
|
-
if
|
|
195
|
+
local = None if targeting else _local_exists(path, q)
|
|
196
|
+
if local:
|
|
197
|
+
_log(f"[bibcite] ({i + 1}/{len(queries)}) {q} — already in file: {local}")
|
|
198
|
+
items.append((q, None, 0, local))
|
|
199
|
+
continue
|
|
200
|
+
if resolved_any:
|
|
132
201
|
time.sleep(1) # one process shares the rate-limit breaker; stay polite
|
|
202
|
+
resolved_any = True
|
|
133
203
|
_log(f"[bibcite] ({i + 1}/{len(queries)}) {q}")
|
|
134
204
|
res, code = _resolve_or_none(q, args.require_published)
|
|
135
|
-
|
|
205
|
+
items.append((q, res, code, ""))
|
|
136
206
|
else:
|
|
137
207
|
if not args.query:
|
|
138
208
|
_log("[bibcite] provide a query (arXiv id / DOI / title), --bibtex, or --from")
|
|
139
209
|
return EXIT_NOT_FOUND
|
|
140
210
|
query = " ".join(args.query)
|
|
211
|
+
local = None if targeting else _local_exists(path, query)
|
|
212
|
+
if local:
|
|
213
|
+
_log(f"[bibcite] already in file (matched locally, no network): {local}")
|
|
214
|
+
_emit({"action": "exists", "key": local, "file": str(path), "tidied": False})
|
|
215
|
+
return 0
|
|
141
216
|
res, code = _resolve_or_none(query, args.require_published)
|
|
142
217
|
if res is None:
|
|
143
218
|
return code
|
|
144
|
-
|
|
219
|
+
items.append((query, res, 0, ""))
|
|
145
220
|
|
|
146
221
|
# Write all entries first, tidy once, then read back the final keys.
|
|
147
222
|
results = []
|
|
148
223
|
wrote = False
|
|
149
|
-
for query, res, code in
|
|
224
|
+
for query, res, code, local_key in items:
|
|
225
|
+
if local_key:
|
|
226
|
+
results.append({"query": query, "action": "exists", "key": local_key})
|
|
227
|
+
continue
|
|
150
228
|
if res is None:
|
|
151
229
|
results.append({"query": query, "action": "failed", "exit_code": code})
|
|
152
230
|
continue
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
"
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
"title": res.entry.get("title", ""),
|
|
161
|
-
"venue": res.venue or "arXiv (preprint)",
|
|
162
|
-
"published": res.published,
|
|
163
|
-
"source": res.source,
|
|
164
|
-
}
|
|
231
|
+
warning = ""
|
|
232
|
+
if args.key:
|
|
233
|
+
warning = _identity_mismatch(path, args.key, res.entry)
|
|
234
|
+
if warning:
|
|
235
|
+
_log(f"[bibcite] warning: {warning}")
|
|
236
|
+
action, key = bibfile.upsert_entry(
|
|
237
|
+
path, res.entry, replace=args.replace, replace_key=args.key or ""
|
|
165
238
|
)
|
|
239
|
+
if action == "no_match_to_replace":
|
|
240
|
+
# A replace that matches nothing must fail loudly, never silently
|
|
241
|
+
# add a duplicate entry.
|
|
242
|
+
_log(
|
|
243
|
+
f"[bibcite] no matching entry to replace for '{query}'"
|
|
244
|
+
+ (f" (key: {args.key})" if args.key else "")
|
|
245
|
+
+ " — nothing written. Use `bibcite add --key <existing-key>` to target one."
|
|
246
|
+
)
|
|
247
|
+
results.append({"query": query, "action": action, "exit_code": EXIT_NOT_FOUND})
|
|
248
|
+
continue
|
|
249
|
+
wrote = wrote or action != "exists"
|
|
250
|
+
result = {
|
|
251
|
+
"query": query,
|
|
252
|
+
"action": action,
|
|
253
|
+
"key": key,
|
|
254
|
+
"title": res.entry.get("title", ""),
|
|
255
|
+
"venue": res.venue or "arXiv (preprint)",
|
|
256
|
+
"published": res.published,
|
|
257
|
+
"source": res.source,
|
|
258
|
+
}
|
|
259
|
+
if warning:
|
|
260
|
+
result["warning"] = warning
|
|
261
|
+
results.append(result)
|
|
166
262
|
|
|
167
263
|
tidied = False
|
|
168
264
|
if wrote and not args.no_tidy:
|
|
@@ -246,6 +342,10 @@ def _upgrade_entries(path: Path, dry_run: bool) -> dict:
|
|
|
246
342
|
entry["year"] = match.year
|
|
247
343
|
if match.doi and not entry.get("doi"):
|
|
248
344
|
entry["doi"] = match.doi
|
|
345
|
+
if match.title:
|
|
346
|
+
# Camera-ready titles drift from arXiv ones; the published
|
|
347
|
+
# title is the correct one to cite.
|
|
348
|
+
entry["title"] = match.title
|
|
249
349
|
changed += 1
|
|
250
350
|
report.append(
|
|
251
351
|
{
|
|
@@ -294,11 +394,30 @@ def _check_problems(path: Path) -> tuple[int, list] | None:
|
|
|
294
394
|
return None
|
|
295
395
|
problems = []
|
|
296
396
|
seen_titles: dict[str, str] = {}
|
|
397
|
+
by_author: dict[str, list[tuple[str, str]]] = {} # lastname -> [(key, title)]
|
|
297
398
|
for entry in db.entries:
|
|
298
399
|
key = entry.get("ID", "?")
|
|
299
400
|
nt = norm_title(entry.get("title", ""))
|
|
300
401
|
if nt and nt in seen_titles:
|
|
301
402
|
problems.append({"key": key, "issue": f"duplicate title of {seen_titles[nt]}"})
|
|
403
|
+
elif nt:
|
|
404
|
+
# Near-duplicates (title drift: same first author, similar title)
|
|
405
|
+
# slip past exact matching — exactly how a failed replace plus a
|
|
406
|
+
# re-add pollutes a file.
|
|
407
|
+
last = (
|
|
408
|
+
first_author_last_name(entry["author"]) if entry.get("author") else ""
|
|
409
|
+
)
|
|
410
|
+
for other_key, other_title in by_author.get(last, []):
|
|
411
|
+
if titles_similar(entry.get("title", ""), other_title):
|
|
412
|
+
problems.append(
|
|
413
|
+
{
|
|
414
|
+
"key": key,
|
|
415
|
+
"issue": f"near-duplicate of {other_key} (title drift?)",
|
|
416
|
+
}
|
|
417
|
+
)
|
|
418
|
+
break
|
|
419
|
+
if last:
|
|
420
|
+
by_author.setdefault(last, []).append((key, entry.get("title", "")))
|
|
302
421
|
seen_titles.setdefault(nt, key)
|
|
303
422
|
for f in ("author", "title", "year"):
|
|
304
423
|
if not entry.get(f):
|
|
@@ -388,7 +507,8 @@ def main(argv=None) -> int:
|
|
|
388
507
|
a.add_argument("query", nargs="*", help="arXiv id / arXiv URL / DOI / title")
|
|
389
508
|
a.add_argument("--bibtex", help="raw BibTeX entry to add instead of a query ('-' reads stdin)")
|
|
390
509
|
a.add_argument("--from", dest="from_file", metavar="FILE", help="batch mode: one query per line (shares rate-limit state, tidies once)")
|
|
391
|
-
a.add_argument("--replace", action="store_true", help="overwrite an existing matching entry (keeps its citation key)")
|
|
510
|
+
a.add_argument("--replace", action="store_true", help="overwrite an existing matching entry (keeps its citation key); errors if nothing matches")
|
|
511
|
+
a.add_argument("--key", metavar="KEY", help="replace exactly the entry with this citation key (for title drift)")
|
|
392
512
|
a.add_argument("--no-tidy", action="store_true")
|
|
393
513
|
a.add_argument("--no-cache", action="store_true", help="bypass the local match cache")
|
|
394
514
|
a.add_argument("--require-published", action="store_true")
|
|
@@ -76,6 +76,30 @@ def first_author_last_name(author_field: str) -> str:
|
|
|
76
76
|
return mini_hash(last) or "anon"
|
|
77
77
|
|
|
78
78
|
|
|
79
|
+
def sig_tokens(title: str) -> set[str]:
|
|
80
|
+
"""Significant title tokens: folded, alphanumeric, stopwords removed."""
|
|
81
|
+
tokens = re.split(r"[^a-z0-9]+", fold_ascii(title).lower())
|
|
82
|
+
return {t for t in tokens if len(t) > 2 and t not in ENGLISH_STOPWORDS}
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def titles_similar(a: str, b: str, threshold: float = 0.75) -> bool:
|
|
86
|
+
"""Token-overlap similarity — catches preprint→camera-ready title drift
|
|
87
|
+
("Information-Theoretic Perspective" vs "Information Theory Perspective")
|
|
88
|
+
without matching genuinely different papers.
|
|
89
|
+
|
|
90
|
+
Uses the overlap coefficient (|∩| / min) rather than Jaccard so one
|
|
91
|
+
changed word in a shortish title still matches; very short titles
|
|
92
|
+
(<=3 significant tokens, e.g. "Deep Learning") must match exactly
|
|
93
|
+
because a single shared word would otherwise dominate."""
|
|
94
|
+
ta, tb = sig_tokens(a), sig_tokens(b)
|
|
95
|
+
if not ta or not tb:
|
|
96
|
+
return False
|
|
97
|
+
smaller = min(len(ta), len(tb))
|
|
98
|
+
if smaller <= 3:
|
|
99
|
+
return ta == tb
|
|
100
|
+
return len(ta & tb) / smaller >= threshold
|
|
101
|
+
|
|
102
|
+
|
|
79
103
|
def fix_author_caps(author_field: str) -> str:
|
|
80
104
|
"""Normalize ALL-CAPS author names (old CrossRef records store e.g.
|
|
81
105
|
"EPPS, T. W. and PULLEY, LAWRENCE B."). A word is re-cased only when it
|
|
@@ -98,6 +122,12 @@ def fix_author_caps(author_field: str) -> str:
|
|
|
98
122
|
return " and ".join(fix_name(n) for n in names)
|
|
99
123
|
|
|
100
124
|
|
|
125
|
+
def fix_pages(pages: str) -> str:
|
|
126
|
+
"""BibTeX page ranges use `--`; CrossRef emits en-dashes (411–430) and
|
|
127
|
+
some sources a single hyphen. Collapse any dash run to `--`."""
|
|
128
|
+
return re.sub(r"\s*[-‐-―]+\s*", "--", pages.strip())
|
|
129
|
+
|
|
130
|
+
|
|
101
131
|
def make_key(author_field: str, year: str | int, title: str) -> str:
|
|
102
132
|
"""Deterministic citation key: <lastname><year><firstword>.
|
|
103
133
|
|
|
@@ -10,7 +10,13 @@ import sys
|
|
|
10
10
|
from dataclasses import dataclass
|
|
11
11
|
|
|
12
12
|
from .bibfile import NOISE_FIELDS, parse_bibtex_entry
|
|
13
|
-
from .normalize import
|
|
13
|
+
from .normalize import (
|
|
14
|
+
clean_title,
|
|
15
|
+
first_author_last_name,
|
|
16
|
+
fix_author_caps,
|
|
17
|
+
fix_pages,
|
|
18
|
+
make_key,
|
|
19
|
+
)
|
|
14
20
|
|
|
15
21
|
|
|
16
22
|
class NotFound(Exception):
|
|
@@ -153,6 +159,8 @@ def _finalize(entry: dict, meta: ArxivMeta | None) -> dict:
|
|
|
153
159
|
# a missing url from the DOI.
|
|
154
160
|
if not url or "dx.doi.org" in url:
|
|
155
161
|
entry["url"] = f"https://doi.org/{entry['doi']}"
|
|
162
|
+
if entry.get("pages"):
|
|
163
|
+
entry["pages"] = fix_pages(entry["pages"])
|
|
156
164
|
author = entry.get("author", "") or "anonymous"
|
|
157
165
|
year = entry.get("year", "") or "XXXX"
|
|
158
166
|
entry["ID"] = make_key(author, year, entry.get("title", ""))
|
|
@@ -16,7 +16,7 @@ from dataclasses import dataclass, field
|
|
|
16
16
|
|
|
17
17
|
import httpx
|
|
18
18
|
|
|
19
|
-
from .normalize import clean_title, mini_hash, norm_title
|
|
19
|
+
from .normalize import clean_title, mini_hash, norm_title, sig_tokens, titles_similar
|
|
20
20
|
|
|
21
21
|
UA = "bibcite/0.1 (https://github.com/leonardo/bibcite; mailto:bibcite@gmail.com)"
|
|
22
22
|
BROWSER_UA = (
|
|
@@ -198,6 +198,73 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
|
|
|
198
198
|
return None
|
|
199
199
|
|
|
200
200
|
|
|
201
|
+
def _dblp_hit_authors(info: dict) -> list[str]:
|
|
202
|
+
authors = (info.get("authors") or {}).get("author") or []
|
|
203
|
+
if isinstance(authors, dict):
|
|
204
|
+
authors = [authors]
|
|
205
|
+
return [a.get("text", "") for a in authors if isinstance(a, dict)]
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None:
|
|
209
|
+
"""Title-drift fallback: camera-ready titles often differ from the arXiv
|
|
210
|
+
ones ("Information-Theoretic" -> "Information Theory"), and DBLP's
|
|
211
|
+
token-AND search then misses entirely. Query author + the most
|
|
212
|
+
distinctive title tokens instead, and accept token-Jaccard-similar
|
|
213
|
+
titles — guarded by author and year so different papers can't sneak in.
|
|
214
|
+
"""
|
|
215
|
+
if not author_hint:
|
|
216
|
+
return None
|
|
217
|
+
tokens = sorted(sig_tokens(title), key=len, reverse=True)[:3]
|
|
218
|
+
if not tokens:
|
|
219
|
+
return None
|
|
220
|
+
q = " ".join([author_hint] + tokens)
|
|
221
|
+
with _client() as c:
|
|
222
|
+
r = c.get(
|
|
223
|
+
"https://dblp.org/search/publ/api",
|
|
224
|
+
params={"q": q, "format": "json", "h": 100},
|
|
225
|
+
)
|
|
226
|
+
if r.status_code == 429:
|
|
227
|
+
raise SourceUnavailable("DBLP rate-limited (429)")
|
|
228
|
+
r.raise_for_status()
|
|
229
|
+
hits = r.json().get("result", {}).get("hits", {}).get("hit", []) or []
|
|
230
|
+
hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
|
|
231
|
+
for hit in hits:
|
|
232
|
+
info = hit.get("info", {})
|
|
233
|
+
hit_title = clean_title(html.unescape(info.get("title", "")))
|
|
234
|
+
if info.get("venue") == "CoRR" or not info.get("venue"):
|
|
235
|
+
continue
|
|
236
|
+
if not titles_similar(hit_title, title):
|
|
237
|
+
continue
|
|
238
|
+
if year and info.get("year"):
|
|
239
|
+
if abs(int(info["year"]) - int(year)) > 2:
|
|
240
|
+
continue
|
|
241
|
+
hit_authors = mini_hash(" ".join(_dblp_hit_authors(info)))
|
|
242
|
+
if author_hint not in hit_authors:
|
|
243
|
+
continue
|
|
244
|
+
venue = info["venue"]
|
|
245
|
+
if isinstance(venue, list):
|
|
246
|
+
venue = venue[0]
|
|
247
|
+
bibtex = ""
|
|
248
|
+
if info.get("url"):
|
|
249
|
+
br = c.get(info["url"] + ".bib")
|
|
250
|
+
if br.status_code == 200:
|
|
251
|
+
bibtex = br.text
|
|
252
|
+
_log(
|
|
253
|
+
f"[dblp-fuzzy] match with title drift: '{hit_title}' "
|
|
254
|
+
f"@ {venue} {info.get('year', '')}"
|
|
255
|
+
)
|
|
256
|
+
return Match(
|
|
257
|
+
source="dblp-fuzzy",
|
|
258
|
+
venue=str(venue),
|
|
259
|
+
title=hit_title,
|
|
260
|
+
year=str(info.get("year", "")),
|
|
261
|
+
doi=info.get("doi", ""),
|
|
262
|
+
bibtex=bibtex,
|
|
263
|
+
url=info.get("ee", "") or info.get("url", ""),
|
|
264
|
+
)
|
|
265
|
+
return None
|
|
266
|
+
|
|
267
|
+
|
|
201
268
|
# ---------------------------------------------------------------------------
|
|
202
269
|
# Semantic Scholar
|
|
203
270
|
# ---------------------------------------------------------------------------
|
|
@@ -627,4 +694,19 @@ def find_published(
|
|
|
627
694
|
_log(f"[{name}] disabled for the rest of this run: {e}")
|
|
628
695
|
except Exception as e: # network hiccup on one source must not kill the run
|
|
629
696
|
_log(f"[{name}] error: {type(e).__name__}: {e}")
|
|
697
|
+
|
|
698
|
+
# Exact-title search missed everywhere. Before concluding "no published
|
|
699
|
+
# version", try the title-drift fallback — camera-ready titles frequently
|
|
700
|
+
# differ from the arXiv ones, which is precisely the upgrade scenario.
|
|
701
|
+
if author_hint and "dblp" not in _DISABLED:
|
|
702
|
+
try:
|
|
703
|
+
m = try_dblp_fuzzy(title, author_hint, year)
|
|
704
|
+
if m:
|
|
705
|
+
cache.put(cache_key, m.__dict__)
|
|
706
|
+
return m, "found"
|
|
707
|
+
clean_misses += 1
|
|
708
|
+
except SourceUnavailable as e:
|
|
709
|
+
_DISABLED["dblp"] = str(e)
|
|
710
|
+
except Exception as e:
|
|
711
|
+
_log(f"[dblp-fuzzy] error: {type(e).__name__}: {e}")
|
|
630
712
|
return None, ("not_found" if clean_misses else "unavailable")
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Regression tests for the second round of field-use bug reports."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from bibcite.bibfile import MONTH_STRINGS, load_bib_file, upsert_entry, _write_db
|
|
6
|
+
from bibcite.normalize import titles_similar
|
|
7
|
+
|
|
8
|
+
ARXIV_TITLE = "An Information-Theoretic Perspective on Variance-Invariance-Covariance Regularization"
|
|
9
|
+
PUBLISHED_TITLE = "An Information Theory Perspective on Variance-Invariance-Covariance Regularization"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_titles_similar_catches_camera_ready_drift():
|
|
13
|
+
assert titles_similar(ARXIV_TITLE, PUBLISHED_TITLE)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_titles_similar_rejects_different_papers():
|
|
17
|
+
assert not titles_similar(
|
|
18
|
+
"Attention Is All You Need",
|
|
19
|
+
"An Image is Worth 16x16 Words: Transformers for Image Recognition",
|
|
20
|
+
)
|
|
21
|
+
assert not titles_similar("Deep Residual Learning", "")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
ENTRY = {
|
|
25
|
+
"ENTRYTYPE": "inproceedings",
|
|
26
|
+
"ID": "k1",
|
|
27
|
+
"title": "Paper One",
|
|
28
|
+
"author": "A B",
|
|
29
|
+
"booktitle": "Some Conference (SC)",
|
|
30
|
+
"year": "2020",
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_replace_without_match_errors_instead_of_adding(tmp_path: Path):
|
|
35
|
+
bib = tmp_path / "r.bib"
|
|
36
|
+
upsert_entry(bib, dict(ENTRY))
|
|
37
|
+
stranger = dict(ENTRY, ID="k2", title="A Totally Different Paper")
|
|
38
|
+
action, key = upsert_entry(bib, stranger, replace=True)
|
|
39
|
+
assert action == "no_match_to_replace"
|
|
40
|
+
assert "Totally Different" not in bib.read_text() # nothing was written
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def test_replace_key_targets_specific_entry(tmp_path: Path):
|
|
44
|
+
bib = tmp_path / "r.bib"
|
|
45
|
+
upsert_entry(bib, dict(ENTRY))
|
|
46
|
+
drifted = dict(ENTRY, ID="whatever", title="Paper One Revised Title")
|
|
47
|
+
action, key = upsert_entry(bib, drifted, replace_key="k1")
|
|
48
|
+
assert (action, key) == ("replaced", "k1")
|
|
49
|
+
assert "Paper One Revised Title" in bib.read_text()
|
|
50
|
+
action, _ = upsert_entry(bib, drifted, replace_key="nonexistent")
|
|
51
|
+
assert action == "no_match_to_replace"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_month_strings_never_written_to_file(tmp_path: Path):
|
|
55
|
+
bib = tmp_path / "m.bib"
|
|
56
|
+
# Simulate a file polluted by the old bug: @string month macros present.
|
|
57
|
+
bib.write_text(
|
|
58
|
+
'@string{january = {January}}\n'
|
|
59
|
+
'@article{x, title = {T}, author = {A B}, year = {2000}, month = january }\n'
|
|
60
|
+
)
|
|
61
|
+
db = load_bib_file(bib)
|
|
62
|
+
_write_db(bib, db)
|
|
63
|
+
text = bib.read_text()
|
|
64
|
+
assert "@string" not in text # scrubbed on write
|
|
65
|
+
assert "title" in text
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_month_strings_cover_all_months():
|
|
69
|
+
for m in ("january", "may", "june", "december", "jan", "jun", "dec"):
|
|
70
|
+
assert m in MONTH_STRINGS
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Regression tests for the third round of field-use reports."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from bibcite.bibfile import (
|
|
6
|
+
_scrub_month_strings,
|
|
7
|
+
find_existing,
|
|
8
|
+
load_bib_file,
|
|
9
|
+
upsert_entry,
|
|
10
|
+
)
|
|
11
|
+
from bibcite.normalize import fix_pages
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_fix_pages_dashes():
|
|
15
|
+
assert fix_pages("411–430") == "411--430" # en-dash
|
|
16
|
+
assert fix_pages("411-430") == "411--430" # single hyphen
|
|
17
|
+
assert fix_pages("411 -- 430") == "411--430"
|
|
18
|
+
assert fix_pages("723—726") == "723--726" # em-dash
|
|
19
|
+
assert fix_pages("e123") == "e123" # no range untouched
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
PREPRINT = {
|
|
23
|
+
"ENTRYTYPE": "misc",
|
|
24
|
+
"ID": "old",
|
|
25
|
+
"title": "An Information-Theoretic Perspective on VICReg",
|
|
26
|
+
"author": "Ravid Shwartz-Ziv and Yann LeCun",
|
|
27
|
+
"howpublished": "arXiv preprint arXiv:2303.00633",
|
|
28
|
+
"eprint": "2303.00633",
|
|
29
|
+
"year": "2023",
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_find_existing_by_doi_in_url(tmp_path: Path):
|
|
34
|
+
bib = tmp_path / "d.bib"
|
|
35
|
+
upsert_entry(
|
|
36
|
+
bib,
|
|
37
|
+
{
|
|
38
|
+
"ENTRYTYPE": "article",
|
|
39
|
+
"ID": "k",
|
|
40
|
+
"title": "T",
|
|
41
|
+
"author": "A B",
|
|
42
|
+
"journal": "J",
|
|
43
|
+
"year": "2000",
|
|
44
|
+
"url": "https://doi.org/10.1093/biomet/70.3.723",
|
|
45
|
+
},
|
|
46
|
+
)
|
|
47
|
+
db = load_bib_file(bib)
|
|
48
|
+
# No doi field on the entry — matched via the url (pre-0.4.0 files).
|
|
49
|
+
assert find_existing(db, "", doi="10.1093/biomet/70.3.723") is not None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_dedupe_catches_title_drift_pair(tmp_path: Path):
|
|
53
|
+
bib = tmp_path / "p.bib"
|
|
54
|
+
upsert_entry(bib, dict(PREPRINT))
|
|
55
|
+
published = {
|
|
56
|
+
"ENTRYTYPE": "inproceedings",
|
|
57
|
+
"ID": "new",
|
|
58
|
+
"title": "An Information Theory Perspective on VICReg", # drifted
|
|
59
|
+
"author": "Ravid Shwartz-Ziv and Yann LeCun",
|
|
60
|
+
"booktitle": "Advances in Neural Information Processing Systems (NeurIPS)",
|
|
61
|
+
"year": "2023",
|
|
62
|
+
}
|
|
63
|
+
action, key = upsert_entry(bib, published)
|
|
64
|
+
# Fuzzy same-author dedupe: upgraded in place, NOT added as a duplicate.
|
|
65
|
+
assert (action, key) == ("upgraded", "old")
|
|
66
|
+
assert bib.read_text().count("@") == 1
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_scrub_orphan_month_strings(tmp_path: Path):
|
|
70
|
+
bib = tmp_path / "m.bib"
|
|
71
|
+
bib.write_text(
|
|
72
|
+
"@string{january = {January}}\n@string{june = {June}}\n"
|
|
73
|
+
"@article{x, title = {T}, author = {A B}, year = {2000} }\n"
|
|
74
|
+
)
|
|
75
|
+
_scrub_month_strings(bib)
|
|
76
|
+
text = bib.read_text()
|
|
77
|
+
assert "@string" not in text
|
|
78
|
+
assert "title" in text
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_scrub_leaves_clean_files_alone(tmp_path: Path):
|
|
82
|
+
bib = tmp_path / "c.bib"
|
|
83
|
+
original = "@article{x,\n title = {T},\n author = {A B},\n year = {2000},\n}\n"
|
|
84
|
+
bib.write_text(original)
|
|
85
|
+
_scrub_month_strings(bib)
|
|
86
|
+
assert bib.read_text() == original # untouched, not even rewritten
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|