bibcite-cli 0.3.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/PKG-INFO +1 -1
  2. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/pyproject.toml +1 -1
  3. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/__init__.py +1 -1
  4. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/bibfile.py +83 -13
  5. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/cli.py +141 -21
  6. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/normalize.py +30 -0
  7. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/resolve.py +9 -1
  8. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/sources.py +83 -1
  9. bibcite_cli-0.4.1/tests/test_round2.py +70 -0
  10. bibcite_cli-0.4.1/tests/test_round3.py +86 -0
  11. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/uv.lock +1 -1
  12. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/.gitignore +0 -0
  13. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/LICENSE +0 -0
  14. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/Readme.md +0 -0
  15. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/cache.py +0 -0
  16. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/data/strings.bib +0 -0
  17. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/src/bibcite/venues.py +0 -0
  18. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_bibfile.py +0 -0
  19. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_bugfixes.py +0 -0
  20. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_entry_types.py +0 -0
  21. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_normalize.py +0 -0
  22. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_strings_override.py +0 -0
  23. {bibcite_cli-0.3.0 → bibcite_cli-0.4.1}/tests/test_venues.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bibcite-cli
3
- Version: 0.3.0
3
+ Version: 0.4.1
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.3.0"
3
+ version = "0.4.1"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """bibcite: canonical BibTeX resolution for papers (arXiv id / DOI / title)."""
2
2
 
3
- __version__ = "0.3.0"
3
+ __version__ = "0.4.1"
@@ -20,14 +20,16 @@ from .normalize import norm_title
20
20
  # \cite{} commands valid.
21
21
  TIDY_ARGS = [
22
22
  "--modify",
23
- "--omit=pages,publisher,doi,timestamp,biburl,bibsource,abstract,month,series,volume,editor,note,date,number,address",
23
+ # volume/number/pages/doi are kept (bibliographic substance the user
24
+ # asked to retain); the omit list drops only true noise.
25
+ "--omit=publisher,timestamp,biburl,bibsource,abstract,month,series,editor,note,date,address",
24
26
  "--curly",
25
27
  "--blank-lines",
26
28
  "--trailing-commas",
27
29
  "--sort=-year",
28
30
  "--duplicates=citation",
29
31
  "--merge=first",
30
- "--sort-fields=author,title,booktitle,journal,year,url,pdf",
32
+ "--sort-fields=author,title,booktitle,journal,volume,number,pages,year,doi,url,pdf",
31
33
  "--strip-enclosing-braces",
32
34
  "--tidy-comments",
33
35
  ]
@@ -131,45 +133,86 @@ def load_bib_file(path: Path) -> BibDatabase | None:
131
133
  return None
132
134
 
133
135
 
134
- def find_existing(db: BibDatabase, title: str, arxiv_id: str = "", doi: str = "") -> dict | None:
136
+ def find_existing(
137
+ db: BibDatabase,
138
+ title: str,
139
+ arxiv_id: str = "",
140
+ doi: str = "",
141
+ author: str = "",
142
+ ) -> dict | None:
143
+ from .normalize import first_author_last_name, titles_similar
144
+
135
145
  ref = norm_title(title)
136
146
  for entry in db.entries:
137
147
  if arxiv_id and entry_arxiv_id(entry) == arxiv_id:
138
148
  return entry
139
- if doi and entry.get("doi", "").lower() == doi.lower():
140
- return entry
149
+ if doi:
150
+ d = doi.lower()
151
+ # Older entries may lack a doi field but carry it in the url.
152
+ if entry.get("doi", "").lower() == d or d in entry.get("url", "").lower():
153
+ return entry
141
154
  if ref and norm_title(entry.get("title", "")) == ref:
142
155
  return entry
156
+ # Fuzzy pass: title drift (arXiv vs camera-ready) with the same first
157
+ # author is the same paper — catch it BEFORE writing a duplicate pair.
158
+ if title and author:
159
+ last = first_author_last_name(author)
160
+ for entry in db.entries:
161
+ if not entry.get("author"):
162
+ continue
163
+ if first_author_last_name(entry["author"]) != last:
164
+ continue
165
+ if titles_similar(title, entry.get("title", "")):
166
+ return entry
143
167
  return None
144
168
 
145
169
 
146
- def upsert_entry(path: Path, entry: dict, replace: bool = False) -> tuple[str, str]:
170
+ def upsert_entry(
171
+ path: Path, entry: dict, replace: bool = False, replace_key: str = ""
172
+ ) -> tuple[str, str]:
147
173
  """Insert or upgrade ``entry`` in ``path``.
148
174
 
149
175
  Returns (action, key), action in "added" | "upgraded" | "exists" |
150
- "replaced". With ``replace``, an existing matching entry is overwritten
151
- (its citation key is kept so existing \\cite{} commands stay valid).
176
+ "replaced" | "no_match_to_replace". With ``replace``, an existing
177
+ matching entry is overwritten; ``replace_key`` targets a specific entry
178
+ by citation key (for when title drift defeats the automatic match). The
179
+ existing key is always kept so \\cite{} commands stay valid. A replace
180
+ that matches nothing is an ERROR, not a silent add — that is how
181
+ duplicate entries sneak into a file.
152
182
  """
153
183
  db = load_bib_file(path)
154
184
  if db is None: # unparseable file: append blindly
185
+ if replace or replace_key:
186
+ return "no_match_to_replace", replace_key or entry["ID"]
155
187
  with path.open("a") as f:
156
188
  f.write("\n" + entry_to_bibtex(entry))
157
189
  return "added", entry["ID"]
158
190
 
159
- existing = find_existing(
160
- db, entry.get("title", ""), entry_arxiv_id(entry), entry.get("doi", "")
161
- )
191
+ if replace_key:
192
+ existing = next((e for e in db.entries if e.get("ID") == replace_key), None)
193
+ else:
194
+ existing = find_existing(
195
+ db,
196
+ entry.get("title", ""),
197
+ entry_arxiv_id(entry),
198
+ entry.get("doi", ""),
199
+ entry.get("author", ""),
200
+ )
201
+
162
202
  if existing is not None:
163
203
  upgrade = is_preprint(existing) and not is_preprint(entry)
164
- if replace or upgrade:
204
+ if replace or replace_key or upgrade:
165
205
  key = existing["ID"]
166
206
  existing.clear()
167
207
  existing.update({k: str(v) for k, v in entry.items() if v})
168
208
  existing["ID"] = key # keep the key the user may already \cite
169
209
  _write_db(path, db)
170
- return ("replaced" if replace else "upgraded"), key
210
+ return ("replaced" if (replace or replace_key) else "upgraded"), key
171
211
  return "exists", existing["ID"]
172
212
 
213
+ if replace or replace_key:
214
+ return "no_match_to_replace", replace_key or entry["ID"]
215
+
173
216
  db.entries.append({k: str(v) for k, v in entry.items() if v})
174
217
  _write_db(path, db)
175
218
  return "added", entry["ID"]
@@ -190,6 +233,12 @@ def remove_entry(path: Path, key: str) -> bool:
190
233
 
191
234
 
192
235
  def _write_db(path: Path, db: BibDatabase):
236
+ # Never write our injected month macros back out as @string blocks (they
237
+ # exist only so parsing month=June doesn't crash); this also scrubs any
238
+ # that leaked into a file before this guard existed. User-defined
239
+ # @strings are untouched.
240
+ for k in MONTH_STRINGS:
241
+ db.strings.pop(k, None)
193
242
  writer = BibTexWriter()
194
243
  writer.indent = " "
195
244
  writer.order_entries_by = None # preserve file order; tidy re-sorts anyway
@@ -209,7 +258,28 @@ def tidy_command() -> list[str] | None:
209
258
  return None
210
259
 
211
260
 
261
+ _MONTH_STRING_BLOCK = re.compile(
262
+ r"@string\s*\{\s*(?:" + "|".join(MONTH_STRINGS) + r")\s*=",
263
+ re.IGNORECASE,
264
+ )
265
+
266
+
267
+ def _scrub_month_strings(path: Path):
268
+ """Remove orphan month @string blocks left by the pre-0.4 leak.
269
+ bibtex-tidy itself preserves @strings, so tidy alone never cleans them."""
270
+ try:
271
+ if not _MONTH_STRING_BLOCK.search(path.read_text()):
272
+ return
273
+ db = load_bib_file(path)
274
+ if db is not None:
275
+ _write_db(path, db) # _write_db drops the injected month macros
276
+ _log("[bibcite] scrubbed leftover month @string blocks")
277
+ except Exception as e:
278
+ _log(f"[bibcite] month-string scrub skipped: {e}")
279
+
280
+
212
281
  def run_tidy(path: Path) -> bool:
282
+ _scrub_month_strings(path)
213
283
  cmd = tidy_command()
214
284
  if cmd is None:
215
285
  _log("[bibcite] bibtex-tidy not found (npm i -g bibtex-tidy); skipping tidy")
@@ -12,7 +12,8 @@ import time
12
12
  from pathlib import Path
13
13
 
14
14
  from . import bibfile, cache
15
- from .normalize import first_author_last_name, norm_title
15
+ from .normalize import first_author_last_name, fix_pages, norm_title, titles_similar
16
+ from .resolve import classify
16
17
  from .resolve import (
17
18
  NotFound,
18
19
  Resolved,
@@ -107,62 +108,157 @@ def _resolve_user_bibtex(text: str) -> Resolved:
107
108
  entry.pop("journal", None)
108
109
  entry["ENTRYTYPE"] = canonical.entry_type
109
110
  entry[canonical.bib_field] = canonical.name
110
- return Resolved(entry, "user-bibtex", canonical.name if canonical else raw_venue, True)
111
+ if entry.get("pages"):
112
+ entry["pages"] = fix_pages(entry["pages"])
113
+ published = not bibfile.is_preprint(entry)
114
+ return Resolved(
115
+ entry,
116
+ "user-bibtex",
117
+ (canonical.name if canonical else raw_venue) if published else "",
118
+ published,
119
+ )
120
+
121
+
122
+ def _identity_mismatch(path: Path, target_key: str, new_entry: dict) -> str:
123
+ """`--key` is a scalpel — warn when the resolved paper does not look like
124
+ the entry it is about to overwrite (no shared arXiv id, DOI, or similar
125
+ title), because the old key would then be citing a different paper."""
126
+ db = bibfile.load_bib_file(path)
127
+ if db is None:
128
+ return ""
129
+ target = next((e for e in db.entries if e.get("ID") == target_key), None)
130
+ if target is None:
131
+ return ""
132
+ old_aid, new_aid = bibfile.entry_arxiv_id(target), bibfile.entry_arxiv_id(new_entry)
133
+ if old_aid and new_aid and old_aid == new_aid:
134
+ return ""
135
+ old_doi, new_doi = target.get("doi", "").lower(), new_entry.get("doi", "").lower()
136
+ if old_doi and new_doi and old_doi == new_doi:
137
+ return ""
138
+ if titles_similar(target.get("title", ""), new_entry.get("title", "")):
139
+ return ""
140
+ return (
141
+ f"replacing '{target_key}' with what looks like a DIFFERENT paper "
142
+ f"('{target.get('title', '')[:50]}' -> '{new_entry.get('title', '')[:50]}'); "
143
+ "the key will no longer describe its contents"
144
+ )
145
+
146
+
147
+ def _local_exists(path: Path, query: str) -> str | None:
148
+ """Local pre-check: if the query is already in the file as a PUBLISHED
149
+ entry, skip the network entirely (makes --from re-runs and repeated adds
150
+ near-instant). Preprints still resolve online — they may be upgradable."""
151
+ db = bibfile.load_bib_file(path)
152
+ if db is None or not db.entries:
153
+ return None
154
+ kind, value = classify(query)
155
+ if kind == "arxiv":
156
+ existing = bibfile.find_existing(db, "", arxiv_id=value)
157
+ elif kind == "doi":
158
+ existing = bibfile.find_existing(db, "", doi=value)
159
+ else:
160
+ existing = bibfile.find_existing(db, value)
161
+ if existing is None:
162
+ return None
163
+ if not bibfile.is_preprint(existing):
164
+ return existing["ID"]
165
+ if existing.get("pubstate", "").strip("{}") == "preprint":
166
+ # Confirmed preprint-only: nothing to upgrade, no reason to go online.
167
+ return existing["ID"]
168
+ return None
111
169
 
112
170
 
113
171
  def cmd_add(args) -> int:
114
172
  path = Path(args.file)
115
173
  if args.no_cache:
116
174
  cache.DISABLED = True
175
+ targeting = args.replace or bool(args.key)
176
+ if args.key and args.from_file:
177
+ _log("[bibcite] --key targets one entry; it cannot be combined with --from")
178
+ return EXIT_NOT_FOUND
117
179
 
118
180
  # Collect the queries for this invocation (single, --bibtex, or --from).
181
+ # Each item: (query, resolved_or_None, exit_code, local_exists_key).
182
+ items: list[tuple[str, Resolved | None, int, str]] = []
119
183
  if args.bibtex:
120
184
  text = sys.stdin.read() if args.bibtex == "-" else args.bibtex
121
185
  try:
122
- resolutions = [("<bibtex>", _resolve_user_bibtex(text), 0)]
186
+ items.append(("<bibtex>", _resolve_user_bibtex(text), 0, ""))
123
187
  except ValueError as e:
124
188
  _log(f"[bibcite] {e}")
125
189
  return EXIT_NOT_FOUND
126
190
  elif args.from_file:
127
191
  lines = Path(args.from_file).read_text().splitlines()
128
192
  queries = [q.strip() for q in lines if q.strip() and not q.strip().startswith("#")]
129
- resolutions = []
193
+ resolved_any = False
130
194
  for i, q in enumerate(queries):
131
- if i:
195
+ local = None if targeting else _local_exists(path, q)
196
+ if local:
197
+ _log(f"[bibcite] ({i + 1}/{len(queries)}) {q} — already in file: {local}")
198
+ items.append((q, None, 0, local))
199
+ continue
200
+ if resolved_any:
132
201
  time.sleep(1) # one process shares the rate-limit breaker; stay polite
202
+ resolved_any = True
133
203
  _log(f"[bibcite] ({i + 1}/{len(queries)}) {q}")
134
204
  res, code = _resolve_or_none(q, args.require_published)
135
- resolutions.append((q, res, code))
205
+ items.append((q, res, code, ""))
136
206
  else:
137
207
  if not args.query:
138
208
  _log("[bibcite] provide a query (arXiv id / DOI / title), --bibtex, or --from")
139
209
  return EXIT_NOT_FOUND
140
210
  query = " ".join(args.query)
211
+ local = None if targeting else _local_exists(path, query)
212
+ if local:
213
+ _log(f"[bibcite] already in file (matched locally, no network): {local}")
214
+ _emit({"action": "exists", "key": local, "file": str(path), "tidied": False})
215
+ return 0
141
216
  res, code = _resolve_or_none(query, args.require_published)
142
217
  if res is None:
143
218
  return code
144
- resolutions = [(query, res, 0)]
219
+ items.append((query, res, 0, ""))
145
220
 
146
221
  # Write all entries first, tidy once, then read back the final keys.
147
222
  results = []
148
223
  wrote = False
149
- for query, res, code in resolutions:
224
+ for query, res, code, local_key in items:
225
+ if local_key:
226
+ results.append({"query": query, "action": "exists", "key": local_key})
227
+ continue
150
228
  if res is None:
151
229
  results.append({"query": query, "action": "failed", "exit_code": code})
152
230
  continue
153
- action, key = bibfile.upsert_entry(path, res.entry, replace=args.replace)
154
- wrote = wrote or action != "exists"
155
- results.append(
156
- {
157
- "query": query,
158
- "action": action,
159
- "key": key,
160
- "title": res.entry.get("title", ""),
161
- "venue": res.venue or "arXiv (preprint)",
162
- "published": res.published,
163
- "source": res.source,
164
- }
231
+ warning = ""
232
+ if args.key:
233
+ warning = _identity_mismatch(path, args.key, res.entry)
234
+ if warning:
235
+ _log(f"[bibcite] warning: {warning}")
236
+ action, key = bibfile.upsert_entry(
237
+ path, res.entry, replace=args.replace, replace_key=args.key or ""
165
238
  )
239
+ if action == "no_match_to_replace":
240
+ # A replace that matches nothing must fail loudly, never silently
241
+ # add a duplicate entry.
242
+ _log(
243
+ f"[bibcite] no matching entry to replace for '{query}'"
244
+ + (f" (key: {args.key})" if args.key else "")
245
+ + " — nothing written. Use `bibcite add --key <existing-key>` to target one."
246
+ )
247
+ results.append({"query": query, "action": action, "exit_code": EXIT_NOT_FOUND})
248
+ continue
249
+ wrote = wrote or action != "exists"
250
+ result = {
251
+ "query": query,
252
+ "action": action,
253
+ "key": key,
254
+ "title": res.entry.get("title", ""),
255
+ "venue": res.venue or "arXiv (preprint)",
256
+ "published": res.published,
257
+ "source": res.source,
258
+ }
259
+ if warning:
260
+ result["warning"] = warning
261
+ results.append(result)
166
262
 
167
263
  tidied = False
168
264
  if wrote and not args.no_tidy:
@@ -246,6 +342,10 @@ def _upgrade_entries(path: Path, dry_run: bool) -> dict:
246
342
  entry["year"] = match.year
247
343
  if match.doi and not entry.get("doi"):
248
344
  entry["doi"] = match.doi
345
+ if match.title:
346
+ # Camera-ready titles drift from arXiv ones; the published
347
+ # title is the correct one to cite.
348
+ entry["title"] = match.title
249
349
  changed += 1
250
350
  report.append(
251
351
  {
@@ -294,11 +394,30 @@ def _check_problems(path: Path) -> tuple[int, list] | None:
294
394
  return None
295
395
  problems = []
296
396
  seen_titles: dict[str, str] = {}
397
+ by_author: dict[str, list[tuple[str, str]]] = {} # lastname -> [(key, title)]
297
398
  for entry in db.entries:
298
399
  key = entry.get("ID", "?")
299
400
  nt = norm_title(entry.get("title", ""))
300
401
  if nt and nt in seen_titles:
301
402
  problems.append({"key": key, "issue": f"duplicate title of {seen_titles[nt]}"})
403
+ elif nt:
404
+ # Near-duplicates (title drift: same first author, similar title)
405
+ # slip past exact matching — exactly how a failed replace plus a
406
+ # re-add pollutes a file.
407
+ last = (
408
+ first_author_last_name(entry["author"]) if entry.get("author") else ""
409
+ )
410
+ for other_key, other_title in by_author.get(last, []):
411
+ if titles_similar(entry.get("title", ""), other_title):
412
+ problems.append(
413
+ {
414
+ "key": key,
415
+ "issue": f"near-duplicate of {other_key} (title drift?)",
416
+ }
417
+ )
418
+ break
419
+ if last:
420
+ by_author.setdefault(last, []).append((key, entry.get("title", "")))
302
421
  seen_titles.setdefault(nt, key)
303
422
  for f in ("author", "title", "year"):
304
423
  if not entry.get(f):
@@ -388,7 +507,8 @@ def main(argv=None) -> int:
388
507
  a.add_argument("query", nargs="*", help="arXiv id / arXiv URL / DOI / title")
389
508
  a.add_argument("--bibtex", help="raw BibTeX entry to add instead of a query ('-' reads stdin)")
390
509
  a.add_argument("--from", dest="from_file", metavar="FILE", help="batch mode: one query per line (shares rate-limit state, tidies once)")
391
- a.add_argument("--replace", action="store_true", help="overwrite an existing matching entry (keeps its citation key)")
510
+ a.add_argument("--replace", action="store_true", help="overwrite an existing matching entry (keeps its citation key); errors if nothing matches")
511
+ a.add_argument("--key", metavar="KEY", help="replace exactly the entry with this citation key (for title drift)")
392
512
  a.add_argument("--no-tidy", action="store_true")
393
513
  a.add_argument("--no-cache", action="store_true", help="bypass the local match cache")
394
514
  a.add_argument("--require-published", action="store_true")
@@ -76,6 +76,30 @@ def first_author_last_name(author_field: str) -> str:
76
76
  return mini_hash(last) or "anon"
77
77
 
78
78
 
79
+ def sig_tokens(title: str) -> set[str]:
80
+ """Significant title tokens: folded, alphanumeric, stopwords removed."""
81
+ tokens = re.split(r"[^a-z0-9]+", fold_ascii(title).lower())
82
+ return {t for t in tokens if len(t) > 2 and t not in ENGLISH_STOPWORDS}
83
+
84
+
85
+ def titles_similar(a: str, b: str, threshold: float = 0.75) -> bool:
86
+ """Token-overlap similarity — catches preprint→camera-ready title drift
87
+ ("Information-Theoretic Perspective" vs "Information Theory Perspective")
88
+ without matching genuinely different papers.
89
+
90
+ Uses the overlap coefficient (|∩| / min) rather than Jaccard so one
91
+ changed word in a shortish title still matches; very short titles
92
+ (<=3 significant tokens, e.g. "Deep Learning") must match exactly
93
+ because a single shared word would otherwise dominate."""
94
+ ta, tb = sig_tokens(a), sig_tokens(b)
95
+ if not ta or not tb:
96
+ return False
97
+ smaller = min(len(ta), len(tb))
98
+ if smaller <= 3:
99
+ return ta == tb
100
+ return len(ta & tb) / smaller >= threshold
101
+
102
+
79
103
  def fix_author_caps(author_field: str) -> str:
80
104
  """Normalize ALL-CAPS author names (old CrossRef records store e.g.
81
105
  "EPPS, T. W. and PULLEY, LAWRENCE B."). A word is re-cased only when it
@@ -98,6 +122,12 @@ def fix_author_caps(author_field: str) -> str:
98
122
  return " and ".join(fix_name(n) for n in names)
99
123
 
100
124
 
125
+ def fix_pages(pages: str) -> str:
126
+ """BibTeX page ranges use `--`; CrossRef emits en-dashes (411–430) and
127
+ some sources a single hyphen. Collapse any dash run to `--`."""
128
+ return re.sub(r"\s*[-‐-―]+\s*", "--", pages.strip())
129
+
130
+
101
131
  def make_key(author_field: str, year: str | int, title: str) -> str:
102
132
  """Deterministic citation key: <lastname><year><firstword>.
103
133
 
@@ -10,7 +10,13 @@ import sys
10
10
  from dataclasses import dataclass
11
11
 
12
12
  from .bibfile import NOISE_FIELDS, parse_bibtex_entry
13
- from .normalize import clean_title, first_author_last_name, fix_author_caps, make_key
13
+ from .normalize import (
14
+ clean_title,
15
+ first_author_last_name,
16
+ fix_author_caps,
17
+ fix_pages,
18
+ make_key,
19
+ )
14
20
 
15
21
 
16
22
  class NotFound(Exception):
@@ -153,6 +159,8 @@ def _finalize(entry: dict, meta: ArxivMeta | None) -> dict:
153
159
  # a missing url from the DOI.
154
160
  if not url or "dx.doi.org" in url:
155
161
  entry["url"] = f"https://doi.org/{entry['doi']}"
162
+ if entry.get("pages"):
163
+ entry["pages"] = fix_pages(entry["pages"])
156
164
  author = entry.get("author", "") or "anonymous"
157
165
  year = entry.get("year", "") or "XXXX"
158
166
  entry["ID"] = make_key(author, year, entry.get("title", ""))
@@ -16,7 +16,7 @@ from dataclasses import dataclass, field
16
16
 
17
17
  import httpx
18
18
 
19
- from .normalize import clean_title, mini_hash, norm_title
19
+ from .normalize import clean_title, mini_hash, norm_title, sig_tokens, titles_similar
20
20
 
21
21
  UA = "bibcite/0.1 (https://github.com/leonardo/bibcite; mailto:bibcite@gmail.com)"
22
22
  BROWSER_UA = (
@@ -198,6 +198,73 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
198
198
  return None
199
199
 
200
200
 
201
+ def _dblp_hit_authors(info: dict) -> list[str]:
202
+ authors = (info.get("authors") or {}).get("author") or []
203
+ if isinstance(authors, dict):
204
+ authors = [authors]
205
+ return [a.get("text", "") for a in authors if isinstance(a, dict)]
206
+
207
+
208
+ def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None:
209
+ """Title-drift fallback: camera-ready titles often differ from the arXiv
210
+ ones ("Information-Theoretic" -> "Information Theory"), and DBLP's
211
+ token-AND search then misses entirely. Query author + the most
212
+ distinctive title tokens instead, and accept token-Jaccard-similar
213
+ titles — guarded by author and year so different papers can't sneak in.
214
+ """
215
+ if not author_hint:
216
+ return None
217
+ tokens = sorted(sig_tokens(title), key=len, reverse=True)[:3]
218
+ if not tokens:
219
+ return None
220
+ q = " ".join([author_hint] + tokens)
221
+ with _client() as c:
222
+ r = c.get(
223
+ "https://dblp.org/search/publ/api",
224
+ params={"q": q, "format": "json", "h": 100},
225
+ )
226
+ if r.status_code == 429:
227
+ raise SourceUnavailable("DBLP rate-limited (429)")
228
+ r.raise_for_status()
229
+ hits = r.json().get("result", {}).get("hits", {}).get("hit", []) or []
230
+ hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
231
+ for hit in hits:
232
+ info = hit.get("info", {})
233
+ hit_title = clean_title(html.unescape(info.get("title", "")))
234
+ if info.get("venue") == "CoRR" or not info.get("venue"):
235
+ continue
236
+ if not titles_similar(hit_title, title):
237
+ continue
238
+ if year and info.get("year"):
239
+ if abs(int(info["year"]) - int(year)) > 2:
240
+ continue
241
+ hit_authors = mini_hash(" ".join(_dblp_hit_authors(info)))
242
+ if author_hint not in hit_authors:
243
+ continue
244
+ venue = info["venue"]
245
+ if isinstance(venue, list):
246
+ venue = venue[0]
247
+ bibtex = ""
248
+ if info.get("url"):
249
+ br = c.get(info["url"] + ".bib")
250
+ if br.status_code == 200:
251
+ bibtex = br.text
252
+ _log(
253
+ f"[dblp-fuzzy] match with title drift: '{hit_title}' "
254
+ f"@ {venue} {info.get('year', '')}"
255
+ )
256
+ return Match(
257
+ source="dblp-fuzzy",
258
+ venue=str(venue),
259
+ title=hit_title,
260
+ year=str(info.get("year", "")),
261
+ doi=info.get("doi", ""),
262
+ bibtex=bibtex,
263
+ url=info.get("ee", "") or info.get("url", ""),
264
+ )
265
+ return None
266
+
267
+
201
268
  # ---------------------------------------------------------------------------
202
269
  # Semantic Scholar
203
270
  # ---------------------------------------------------------------------------
@@ -627,4 +694,19 @@ def find_published(
627
694
  _log(f"[{name}] disabled for the rest of this run: {e}")
628
695
  except Exception as e: # network hiccup on one source must not kill the run
629
696
  _log(f"[{name}] error: {type(e).__name__}: {e}")
697
+
698
+ # Exact-title search missed everywhere. Before concluding "no published
699
+ # version", try the title-drift fallback — camera-ready titles frequently
700
+ # differ from the arXiv ones, which is precisely the upgrade scenario.
701
+ if author_hint and "dblp" not in _DISABLED:
702
+ try:
703
+ m = try_dblp_fuzzy(title, author_hint, year)
704
+ if m:
705
+ cache.put(cache_key, m.__dict__)
706
+ return m, "found"
707
+ clean_misses += 1
708
+ except SourceUnavailable as e:
709
+ _DISABLED["dblp"] = str(e)
710
+ except Exception as e:
711
+ _log(f"[dblp-fuzzy] error: {type(e).__name__}: {e}")
630
712
  return None, ("not_found" if clean_misses else "unavailable")
@@ -0,0 +1,70 @@
1
+ """Regression tests for the second round of field-use bug reports."""
2
+
3
+ from pathlib import Path
4
+
5
+ from bibcite.bibfile import MONTH_STRINGS, load_bib_file, upsert_entry, _write_db
6
+ from bibcite.normalize import titles_similar
7
+
8
+ ARXIV_TITLE = "An Information-Theoretic Perspective on Variance-Invariance-Covariance Regularization"
9
+ PUBLISHED_TITLE = "An Information Theory Perspective on Variance-Invariance-Covariance Regularization"
10
+
11
+
12
+ def test_titles_similar_catches_camera_ready_drift():
13
+ assert titles_similar(ARXIV_TITLE, PUBLISHED_TITLE)
14
+
15
+
16
+ def test_titles_similar_rejects_different_papers():
17
+ assert not titles_similar(
18
+ "Attention Is All You Need",
19
+ "An Image is Worth 16x16 Words: Transformers for Image Recognition",
20
+ )
21
+ assert not titles_similar("Deep Residual Learning", "")
22
+
23
+
24
+ ENTRY = {
25
+ "ENTRYTYPE": "inproceedings",
26
+ "ID": "k1",
27
+ "title": "Paper One",
28
+ "author": "A B",
29
+ "booktitle": "Some Conference (SC)",
30
+ "year": "2020",
31
+ }
32
+
33
+
34
+ def test_replace_without_match_errors_instead_of_adding(tmp_path: Path):
35
+ bib = tmp_path / "r.bib"
36
+ upsert_entry(bib, dict(ENTRY))
37
+ stranger = dict(ENTRY, ID="k2", title="A Totally Different Paper")
38
+ action, key = upsert_entry(bib, stranger, replace=True)
39
+ assert action == "no_match_to_replace"
40
+ assert "Totally Different" not in bib.read_text() # nothing was written
41
+
42
+
43
+ def test_replace_key_targets_specific_entry(tmp_path: Path):
44
+ bib = tmp_path / "r.bib"
45
+ upsert_entry(bib, dict(ENTRY))
46
+ drifted = dict(ENTRY, ID="whatever", title="Paper One Revised Title")
47
+ action, key = upsert_entry(bib, drifted, replace_key="k1")
48
+ assert (action, key) == ("replaced", "k1")
49
+ assert "Paper One Revised Title" in bib.read_text()
50
+ action, _ = upsert_entry(bib, drifted, replace_key="nonexistent")
51
+ assert action == "no_match_to_replace"
52
+
53
+
54
+ def test_month_strings_never_written_to_file(tmp_path: Path):
55
+ bib = tmp_path / "m.bib"
56
+ # Simulate a file polluted by the old bug: @string month macros present.
57
+ bib.write_text(
58
+ '@string{january = {January}}\n'
59
+ '@article{x, title = {T}, author = {A B}, year = {2000}, month = january }\n'
60
+ )
61
+ db = load_bib_file(bib)
62
+ _write_db(bib, db)
63
+ text = bib.read_text()
64
+ assert "@string" not in text # scrubbed on write
65
+ assert "title" in text
66
+
67
+
68
+ def test_month_strings_cover_all_months():
69
+ for m in ("january", "may", "june", "december", "jan", "jun", "dec"):
70
+ assert m in MONTH_STRINGS
@@ -0,0 +1,86 @@
1
+ """Regression tests for the third round of field-use reports."""
2
+
3
+ from pathlib import Path
4
+
5
+ from bibcite.bibfile import (
6
+ _scrub_month_strings,
7
+ find_existing,
8
+ load_bib_file,
9
+ upsert_entry,
10
+ )
11
+ from bibcite.normalize import fix_pages
12
+
13
+
14
+ def test_fix_pages_dashes():
15
+ assert fix_pages("411–430") == "411--430" # en-dash
16
+ assert fix_pages("411-430") == "411--430" # single hyphen
17
+ assert fix_pages("411 -- 430") == "411--430"
18
+ assert fix_pages("723—726") == "723--726" # em-dash
19
+ assert fix_pages("e123") == "e123" # no range untouched
20
+
21
+
22
+ PREPRINT = {
23
+ "ENTRYTYPE": "misc",
24
+ "ID": "old",
25
+ "title": "An Information-Theoretic Perspective on VICReg",
26
+ "author": "Ravid Shwartz-Ziv and Yann LeCun",
27
+ "howpublished": "arXiv preprint arXiv:2303.00633",
28
+ "eprint": "2303.00633",
29
+ "year": "2023",
30
+ }
31
+
32
+
33
+ def test_find_existing_by_doi_in_url(tmp_path: Path):
34
+ bib = tmp_path / "d.bib"
35
+ upsert_entry(
36
+ bib,
37
+ {
38
+ "ENTRYTYPE": "article",
39
+ "ID": "k",
40
+ "title": "T",
41
+ "author": "A B",
42
+ "journal": "J",
43
+ "year": "2000",
44
+ "url": "https://doi.org/10.1093/biomet/70.3.723",
45
+ },
46
+ )
47
+ db = load_bib_file(bib)
48
+ # No doi field on the entry — matched via the url (pre-0.4.0 files).
49
+ assert find_existing(db, "", doi="10.1093/biomet/70.3.723") is not None
50
+
51
+
52
+ def test_dedupe_catches_title_drift_pair(tmp_path: Path):
53
+ bib = tmp_path / "p.bib"
54
+ upsert_entry(bib, dict(PREPRINT))
55
+ published = {
56
+ "ENTRYTYPE": "inproceedings",
57
+ "ID": "new",
58
+ "title": "An Information Theory Perspective on VICReg", # drifted
59
+ "author": "Ravid Shwartz-Ziv and Yann LeCun",
60
+ "booktitle": "Advances in Neural Information Processing Systems (NeurIPS)",
61
+ "year": "2023",
62
+ }
63
+ action, key = upsert_entry(bib, published)
64
+ # Fuzzy same-author dedupe: upgraded in place, NOT added as a duplicate.
65
+ assert (action, key) == ("upgraded", "old")
66
+ assert bib.read_text().count("@") == 1
67
+
68
+
69
+ def test_scrub_orphan_month_strings(tmp_path: Path):
70
+ bib = tmp_path / "m.bib"
71
+ bib.write_text(
72
+ "@string{january = {January}}\n@string{june = {June}}\n"
73
+ "@article{x, title = {T}, author = {A B}, year = {2000} }\n"
74
+ )
75
+ _scrub_month_strings(bib)
76
+ text = bib.read_text()
77
+ assert "@string" not in text
78
+ assert "title" in text
79
+
80
+
81
+ def test_scrub_leaves_clean_files_alone(tmp_path: Path):
82
+ bib = tmp_path / "c.bib"
83
+ original = "@article{x,\n title = {T},\n author = {A B},\n year = {2000},\n}\n"
84
+ bib.write_text(original)
85
+ _scrub_month_strings(bib)
86
+ assert bib.read_text() == original # untouched, not even rewritten
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.3.0"
21
+ version = "0.4.1"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes
File without changes