bibcite-cli 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/PKG-INFO +1 -1
  2. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/pyproject.toml +1 -1
  3. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/__init__.py +1 -1
  4. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/bibfile.py +51 -4
  5. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/cli.py +73 -28
  6. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/normalize.py +18 -4
  7. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/resolve.py +33 -5
  8. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/sources.py +73 -17
  9. bibcite_cli-0.5.0/tests/test_round3.py +86 -0
  10. bibcite_cli-0.5.0/tests/test_status_semantics.py +78 -0
  11. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/uv.lock +1 -1
  12. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/.gitignore +0 -0
  13. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/LICENSE +0 -0
  14. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/Readme.md +0 -0
  15. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/cache.py +0 -0
  16. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/data/strings.bib +0 -0
  17. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/src/bibcite/venues.py +0 -0
  18. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_bibfile.py +0 -0
  19. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_bugfixes.py +0 -0
  20. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_entry_types.py +0 -0
  21. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_normalize.py +0 -0
  22. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_round2.py +0 -0
  23. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_strings_override.py +0 -0
  24. {bibcite_cli-0.4.0 → bibcite_cli-0.5.0}/tests/test_venues.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bibcite-cli
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.4.0"
3
+ version = "0.5.0"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """bibcite: canonical BibTeX resolution for papers (arXiv id / DOI / title)."""
2
2
 
3
- __version__ = "0.4.0"
3
+ __version__ = "0.5.0"
@@ -133,15 +133,37 @@ def load_bib_file(path: Path) -> BibDatabase | None:
133
133
  return None
134
134
 
135
135
 
136
- def find_existing(db: BibDatabase, title: str, arxiv_id: str = "", doi: str = "") -> dict | None:
136
+ def find_existing(
137
+ db: BibDatabase,
138
+ title: str,
139
+ arxiv_id: str = "",
140
+ doi: str = "",
141
+ author: str = "",
142
+ ) -> dict | None:
143
+ from .normalize import first_author_last_name, titles_similar
144
+
137
145
  ref = norm_title(title)
138
146
  for entry in db.entries:
139
147
  if arxiv_id and entry_arxiv_id(entry) == arxiv_id:
140
148
  return entry
141
- if doi and entry.get("doi", "").lower() == doi.lower():
142
- return entry
149
+ if doi:
150
+ d = doi.lower()
151
+ # Older entries may lack a doi field but carry it in the url.
152
+ if entry.get("doi", "").lower() == d or d in entry.get("url", "").lower():
153
+ return entry
143
154
  if ref and norm_title(entry.get("title", "")) == ref:
144
155
  return entry
156
+ # Fuzzy pass: title drift (arXiv vs camera-ready) with the same first
157
+ # author is the same paper — catch it BEFORE writing a duplicate pair.
158
+ if title and author:
159
+ last = first_author_last_name(author)
160
+ for entry in db.entries:
161
+ if not entry.get("author"):
162
+ continue
163
+ if first_author_last_name(entry["author"]) != last:
164
+ continue
165
+ if titles_similar(title, entry.get("title", "")):
166
+ return entry
145
167
  return None
146
168
 
147
169
 
@@ -170,7 +192,11 @@ def upsert_entry(
170
192
  existing = next((e for e in db.entries if e.get("ID") == replace_key), None)
171
193
  else:
172
194
  existing = find_existing(
173
- db, entry.get("title", ""), entry_arxiv_id(entry), entry.get("doi", "")
195
+ db,
196
+ entry.get("title", ""),
197
+ entry_arxiv_id(entry),
198
+ entry.get("doi", ""),
199
+ entry.get("author", ""),
174
200
  )
175
201
 
176
202
  if existing is not None:
@@ -232,7 +258,28 @@ def tidy_command() -> list[str] | None:
232
258
  return None
233
259
 
234
260
 
261
+ _MONTH_STRING_BLOCK = re.compile(
262
+ r"@string\s*\{\s*(?:" + "|".join(MONTH_STRINGS) + r")\s*=",
263
+ re.IGNORECASE,
264
+ )
265
+
266
+
267
+ def _scrub_month_strings(path: Path):
268
+ """Remove orphan month @string blocks left by the pre-0.4 leak.
269
+ bibtex-tidy itself preserves @strings, so tidy alone never cleans them."""
270
+ try:
271
+ if not _MONTH_STRING_BLOCK.search(path.read_text()):
272
+ return
273
+ db = load_bib_file(path)
274
+ if db is not None:
275
+ _write_db(path, db) # _write_db drops the injected month macros
276
+ _log("[bibcite] scrubbed leftover month @string blocks")
277
+ except Exception as e:
278
+ _log(f"[bibcite] month-string scrub skipped: {e}")
279
+
280
+
235
281
  def run_tidy(path: Path) -> bool:
282
+ _scrub_month_strings(path)
236
283
  cmd = tidy_command()
237
284
  if cmd is None:
238
285
  _log("[bibcite] bibtex-tidy not found (npm i -g bibtex-tidy); skipping tidy")
@@ -12,7 +12,7 @@ import time
12
12
  from pathlib import Path
13
13
 
14
14
  from . import bibfile, cache
15
- from .normalize import first_author_last_name, norm_title, titles_similar
15
+ from .normalize import first_author_last_name, fix_pages, norm_title, titles_similar
16
16
  from .resolve import classify
17
17
  from .resolve import (
18
18
  NotFound,
@@ -80,18 +80,18 @@ def cmd_get(args) -> int:
80
80
  res, code = _resolve_or_none(query, args.require_published)
81
81
  if res is None:
82
82
  return code
83
- _emit(
84
- {
85
- "action": "resolved",
86
- "key": res.entry["ID"],
87
- "title": res.entry.get("title", ""),
88
- "venue": res.venue or "arXiv (preprint, no published venue found)",
89
- "published": res.published,
90
- "source": res.source,
91
- "bibtex": res.bibtex,
92
- },
93
- args.json,
94
- )
83
+ payload = {
84
+ "action": "resolved",
85
+ "key": res.entry["ID"],
86
+ "title": res.entry.get("title", ""),
87
+ "venue": res.venue or "arXiv (preprint, no published venue found)",
88
+ "published": res.published,
89
+ "source": res.source,
90
+ "bibtex": res.bibtex,
91
+ }
92
+ if not res.published:
93
+ payload["published_check"] = res.check
94
+ _emit(payload, args.json)
95
95
  return 0
96
96
 
97
97
 
@@ -108,6 +108,8 @@ def _resolve_user_bibtex(text: str) -> Resolved:
108
108
  entry.pop("journal", None)
109
109
  entry["ENTRYTYPE"] = canonical.entry_type
110
110
  entry[canonical.bib_field] = canonical.name
111
+ if entry.get("pages"):
112
+ entry["pages"] = fix_pages(entry["pages"])
111
113
  published = not bibfile.is_preprint(entry)
112
114
  return Resolved(
113
115
  entry,
@@ -117,6 +119,31 @@ def _resolve_user_bibtex(text: str) -> Resolved:
117
119
  )
118
120
 
119
121
 
122
+ def _identity_mismatch(path: Path, target_key: str, new_entry: dict) -> str:
123
+ """`--key` is a scalpel — warn when the resolved paper does not look like
124
+ the entry it is about to overwrite (no shared arXiv id, DOI, or similar
125
+ title), because the old key would then be citing a different paper."""
126
+ db = bibfile.load_bib_file(path)
127
+ if db is None:
128
+ return ""
129
+ target = next((e for e in db.entries if e.get("ID") == target_key), None)
130
+ if target is None:
131
+ return ""
132
+ old_aid, new_aid = bibfile.entry_arxiv_id(target), bibfile.entry_arxiv_id(new_entry)
133
+ if old_aid and new_aid and old_aid == new_aid:
134
+ return ""
135
+ old_doi, new_doi = target.get("doi", "").lower(), new_entry.get("doi", "").lower()
136
+ if old_doi and new_doi and old_doi == new_doi:
137
+ return ""
138
+ if titles_similar(target.get("title", ""), new_entry.get("title", "")):
139
+ return ""
140
+ return (
141
+ f"replacing '{target_key}' with what looks like a DIFFERENT paper "
142
+ f"('{target.get('title', '')[:50]}' -> '{new_entry.get('title', '')[:50]}'); "
143
+ "the key will no longer describe its contents"
144
+ )
145
+
146
+
120
147
  def _local_exists(path: Path, query: str) -> str | None:
121
148
  """Local pre-check: if the query is already in the file as a PUBLISHED
122
149
  entry, skip the network entirely (makes --from re-runs and repeated adds
@@ -131,7 +158,12 @@ def _local_exists(path: Path, query: str) -> str | None:
131
158
  existing = bibfile.find_existing(db, "", doi=value)
132
159
  else:
133
160
  existing = bibfile.find_existing(db, value)
134
- if existing is not None and not bibfile.is_preprint(existing):
161
+ if existing is None:
162
+ return None
163
+ if not bibfile.is_preprint(existing):
164
+ return existing["ID"]
165
+ if existing.get("pubstate", "").strip("{}") == "preprint":
166
+ # Confirmed preprint-only: nothing to upgrade, no reason to go online.
135
167
  return existing["ID"]
136
168
  return None
137
169
 
@@ -196,6 +228,11 @@ def cmd_add(args) -> int:
196
228
  if res is None:
197
229
  results.append({"query": query, "action": "failed", "exit_code": code})
198
230
  continue
231
+ warning = ""
232
+ if args.key:
233
+ warning = _identity_mismatch(path, args.key, res.entry)
234
+ if warning:
235
+ _log(f"[bibcite] warning: {warning}")
199
236
  action, key = bibfile.upsert_entry(
200
237
  path, res.entry, replace=args.replace, replace_key=args.key or ""
201
238
  )
@@ -210,17 +247,22 @@ def cmd_add(args) -> int:
210
247
  results.append({"query": query, "action": action, "exit_code": EXIT_NOT_FOUND})
211
248
  continue
212
249
  wrote = wrote or action != "exists"
213
- results.append(
214
- {
215
- "query": query,
216
- "action": action,
217
- "key": key,
218
- "title": res.entry.get("title", ""),
219
- "venue": res.venue or "arXiv (preprint)",
220
- "published": res.published,
221
- "source": res.source,
222
- }
223
- )
250
+ result = {
251
+ "query": query,
252
+ "action": action,
253
+ "key": key,
254
+ "title": res.entry.get("title", ""),
255
+ "venue": res.venue or "arXiv (preprint)",
256
+ "published": res.published,
257
+ "source": res.source,
258
+ }
259
+ if not res.published:
260
+ # "incomplete" = the check ran while core sources were down; the
261
+ # paper may well be published — retry later.
262
+ result["published_check"] = res.check
263
+ if warning:
264
+ result["warning"] = warning
265
+ results.append(result)
224
266
 
225
267
  tidied = False
226
268
  if wrote and not args.no_tidy:
@@ -274,10 +316,13 @@ def _upgrade_entries(path: Path, dry_run: bool) -> dict:
274
316
  )
275
317
  match, status = find_published(title, entry.get("year", ""), aid, hint)
276
318
  if not match:
277
- # "no_published_version" is a trustworthy miss; "sources_unavailable"
278
- # means the sources were down — do not conclude anything.
319
+ # "no_published_version" is trustworthy ONLY when every core
320
+ # source answered; a batch that tripped DBLP's rate limit gets
321
+ # "sources_unavailable" so nobody believes a poisoned verdict.
279
322
  reason = (
280
- "sources_unavailable" if status == "unavailable" else "no_published_version"
323
+ "no_published_version"
324
+ if status == "not_found"
325
+ else "sources_unavailable"
281
326
  )
282
327
  report.append(
283
328
  {"key": entry["ID"], "title": title, "matched": False, "reason": reason}
@@ -82,14 +82,22 @@ def sig_tokens(title: str) -> set[str]:
82
82
  return {t for t in tokens if len(t) > 2 and t not in ENGLISH_STOPWORDS}
83
83
 
84
84
 
85
- def titles_similar(a: str, b: str, threshold: float = 0.7) -> bool:
86
- """Token-Jaccard similarity — catches preprint→camera-ready title drift
85
+ def titles_similar(a: str, b: str, threshold: float = 0.75) -> bool:
86
+ """Token-overlap similarity — catches preprint→camera-ready title drift
87
87
  ("Information-Theoretic Perspective" vs "Information Theory Perspective")
88
- without matching genuinely different papers."""
88
+ without matching genuinely different papers.
89
+
90
+ Uses the overlap coefficient (|∩| / min) rather than Jaccard so one
91
+ changed word in a shortish title still matches; very short titles
92
+ (<=3 significant tokens, e.g. "Deep Learning") must match exactly
93
+ because a single shared word would otherwise dominate."""
89
94
  ta, tb = sig_tokens(a), sig_tokens(b)
90
95
  if not ta or not tb:
91
96
  return False
92
- return len(ta & tb) / len(ta | tb) >= threshold
97
+ smaller = min(len(ta), len(tb))
98
+ if smaller <= 3:
99
+ return ta == tb
100
+ return len(ta & tb) / smaller >= threshold
93
101
 
94
102
 
95
103
  def fix_author_caps(author_field: str) -> str:
@@ -114,6 +122,12 @@ def fix_author_caps(author_field: str) -> str:
114
122
  return " and ".join(fix_name(n) for n in names)
115
123
 
116
124
 
125
+ def fix_pages(pages: str) -> str:
126
+ """BibTeX page ranges use `--`; CrossRef emits en-dashes (411–430) and
127
+ some sources a single hyphen. Collapse any dash run to `--`."""
128
+ return re.sub(r"\s*[-‐-―]+\s*", "--", pages.strip())
129
+
130
+
117
131
  def make_key(author_field: str, year: str | int, title: str) -> str:
118
132
  """Deterministic citation key: <lastname><year><firstword>.
119
133
 
@@ -10,7 +10,13 @@ import sys
10
10
  from dataclasses import dataclass
11
11
 
12
12
  from .bibfile import NOISE_FIELDS, parse_bibtex_entry
13
- from .normalize import clean_title, first_author_last_name, fix_author_caps, make_key
13
+ from .normalize import (
14
+ clean_title,
15
+ first_author_last_name,
16
+ fix_author_caps,
17
+ fix_pages,
18
+ make_key,
19
+ )
14
20
 
15
21
 
16
22
  class NotFound(Exception):
@@ -66,6 +72,12 @@ class Resolved:
66
72
  source: str # where the publication info came from
67
73
  venue: str # final venue string ("" if preprint)
68
74
  published: bool
75
+ # For preprint fallbacks: was the publication check trustworthy?
76
+ # "complete" — every core source answered; the paper really has no
77
+ # published version (as of now)
78
+ # "incomplete" — core sources were rate-limited/down; retry later before
79
+ # believing the preprint status
80
+ check: str = "complete"
69
81
 
70
82
  @property
71
83
  def bibtex(self) -> str:
@@ -153,6 +165,8 @@ def _finalize(entry: dict, meta: ArxivMeta | None) -> dict:
153
165
  # a missing url from the DOI.
154
166
  if not url or "dx.doi.org" in url:
155
167
  entry["url"] = f"https://doi.org/{entry['doi']}"
168
+ if entry.get("pages"):
169
+ entry["pages"] = fix_pages(entry["pages"])
156
170
  author = entry.get("author", "") or "anonymous"
157
171
  year = entry.get("year", "") or "XXXX"
158
172
  entry["ID"] = make_key(author, year, entry.get("title", ""))
@@ -212,9 +226,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
212
226
  f"Could not check publication status for arXiv:{value} (sources down)"
213
227
  )
214
228
  raise NotFound(f"No published version found for arXiv:{value}")
215
- _log("[bibcite] no published version found; using arXiv preprint entry")
229
+ check = "complete" if status == "not_found" else "incomplete"
230
+ if check == "incomplete":
231
+ _log(
232
+ "[bibcite] preprint fallback with INCOMPLETE publication check "
233
+ "(core sources were unavailable) — retry later"
234
+ )
235
+ else:
236
+ _log("[bibcite] no published version found; using arXiv preprint entry")
216
237
  entry = _arxiv_only_entry(meta)
217
- return Resolved(_finalize(entry, meta), "arxiv", "", False)
238
+ return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
218
239
 
219
240
  if kind == "doi":
220
241
  match = crossref_by_doi(value)
@@ -248,9 +269,16 @@ def resolve(query: str, require_published: bool = False) -> Resolved:
248
269
  if meta and meta.arxiv_id:
249
270
  if require_published:
250
271
  raise NotFound(f"Only an arXiv preprint was found for: {value}")
251
- _log("[bibcite] no published version found; using arXiv preprint entry")
272
+ check = "complete" if status == "not_found" else "incomplete"
273
+ if check == "incomplete":
274
+ _log(
275
+ "[bibcite] preprint fallback with INCOMPLETE publication check "
276
+ "(core sources were unavailable) — retry later"
277
+ )
278
+ else:
279
+ _log("[bibcite] no published version found; using arXiv preprint entry")
252
280
  entry = _arxiv_only_entry(meta)
253
- return Resolved(_finalize(entry, meta), "arxiv", "", False)
281
+ return Resolved(_finalize(entry, meta), "arxiv", "", False, check)
254
282
  if status == "unavailable":
255
283
  raise SourcesUnavailable(
256
284
  f"All sources were rate-limited or down while resolving: {value}"
@@ -147,6 +147,42 @@ def arxiv_metadata(arxiv_id: str) -> ArxivMeta:
147
147
  # DBLP
148
148
  # ---------------------------------------------------------------------------
149
149
 
150
+ # DBLP throttles at roughly 1-2 req/s and escalates to temporary IP bans when
151
+ # hammered. Client-side pacing prevents the 429 in the first place; on a 429
152
+ # we back off and retry (honoring Retry-After) instead of instantly poisoning
153
+ # the rest of a batch run — only repeated failure disables the source.
154
+ _DBLP_MIN_INTERVAL = 0.8
155
+ _dblp_last_request = 0.0
156
+
157
+
158
+ def _dblp_get(c: httpx.Client, url: str, params: dict | None = None) -> httpx.Response:
159
+ global _dblp_last_request
160
+ for attempt in range(3):
161
+ wait = _DBLP_MIN_INTERVAL - (time.monotonic() - _dblp_last_request)
162
+ if wait > 0:
163
+ time.sleep(wait)
164
+ _dblp_last_request = time.monotonic()
165
+ try:
166
+ r = c.get(url, params=params)
167
+ except httpx.HTTPError as e: # TCP reset = temporary ban; retrying fast makes it worse
168
+ if attempt < 2:
169
+ time.sleep(5 * (attempt + 1))
170
+ continue
171
+ raise SourceUnavailable(f"DBLP unreachable ({type(e).__name__})")
172
+ if r.status_code == 429:
173
+ retry_after = int(r.headers.get("Retry-After") or 0)
174
+ if retry_after > 30:
175
+ raise SourceUnavailable(f"DBLP rate-limited (Retry-After {retry_after}s)")
176
+ if attempt < 2:
177
+ delay = max(retry_after, 4 * (attempt + 1))
178
+ _log(f"[dblp] 429 — backing off {delay}s")
179
+ time.sleep(delay)
180
+ continue
181
+ raise SourceUnavailable("DBLP rate-limited (429) after backoff retries")
182
+ return r
183
+ raise SourceUnavailable("DBLP unavailable")
184
+
185
+
150
186
  def try_dblp(title: str, author_hint: str = "") -> Match | None:
151
187
  """DBLP search. Generic titles ("X is all you need") drown in DBLP's
152
188
  ranking, so when we know the first author we query with their last name
@@ -157,12 +193,11 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
157
193
  queries.append(title)
158
194
  with _client() as c:
159
195
  for q in queries:
160
- r = c.get(
196
+ r = _dblp_get(
197
+ c,
161
198
  "https://dblp.org/search/publ/api",
162
199
  params={"q": q, "format": "json", "h": 100},
163
200
  )
164
- if r.status_code == 429:
165
- raise SourceUnavailable("DBLP rate-limited (429)")
166
201
  r.raise_for_status()
167
202
  hits = (
168
203
  r.json().get("result", {}).get("hits", {}).get("hit", []) or []
@@ -182,9 +217,12 @@ def try_dblp(title: str, author_hint: str = "") -> Match | None:
182
217
  venue = venue[0]
183
218
  bibtex = ""
184
219
  if info.get("url"):
185
- br = c.get(info["url"] + ".bib")
186
- if br.status_code == 200:
187
- bibtex = br.text
220
+ try:
221
+ br = _dblp_get(c, info["url"] + ".bib")
222
+ if br.status_code == 200:
223
+ bibtex = br.text
224
+ except SourceUnavailable:
225
+ pass # keep the match; construct from fields
188
226
  _log(f"[dblp] match: {venue} {info.get('year', '')}")
189
227
  return Match(
190
228
  source="dblp",
@@ -219,12 +257,11 @@ def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None
219
257
  return None
220
258
  q = " ".join([author_hint] + tokens)
221
259
  with _client() as c:
222
- r = c.get(
260
+ r = _dblp_get(
261
+ c,
223
262
  "https://dblp.org/search/publ/api",
224
263
  params={"q": q, "format": "json", "h": 100},
225
264
  )
226
- if r.status_code == 429:
227
- raise SourceUnavailable("DBLP rate-limited (429)")
228
265
  r.raise_for_status()
229
266
  hits = r.json().get("result", {}).get("hits", {}).get("hit", []) or []
230
267
  hits.sort(key=lambda h: int(h.get("info", {}).get("year", 9999)))
@@ -246,9 +283,12 @@ def try_dblp_fuzzy(title: str, author_hint: str, year: str = "") -> Match | None
246
283
  venue = venue[0]
247
284
  bibtex = ""
248
285
  if info.get("url"):
249
- br = c.get(info["url"] + ".bib")
250
- if br.status_code == 200:
251
- bibtex = br.text
286
+ try:
287
+ br = _dblp_get(c, info["url"] + ".bib")
288
+ if br.status_code == 200:
289
+ bibtex = br.text
290
+ except SourceUnavailable:
291
+ pass # keep the match; construct from fields
252
292
  _log(
253
293
  f"[dblp-fuzzy] match with title drift: '{hit_title}' "
254
294
  f"@ {venue} {info.get('year', '')}"
@@ -659,16 +699,24 @@ CASCADE = (
659
699
  # (PaperMemory's DISABLE_MATCH, ported).
660
700
  _DISABLED: dict[str, str] = {}
661
701
 
702
+ # Only these sources are authoritative enough that losing one taints a miss
703
+ # into "incomplete". Google Scholar captchas and Unpaywall flakiness are
704
+ # routine and must not stop "not_found" from ever being trustworthy.
705
+ CORE_SOURCES = frozenset({"dblp", "semanticscholar", "crossref", "openalex"})
706
+
662
707
 
663
708
  def find_published(
664
709
  title: str, year: str = "", arxiv_id: str = "", author_hint: str = ""
665
710
  ) -> tuple[Match | None, str]:
666
711
  """Try each source in order; first verified hit wins.
667
712
 
668
- Returns (match, status). status distinguishes a trustworthy miss from an
669
- outage: "found" | "not_found" (>=1 source answered cleanly with no hit) |
670
- "unavailable" (every source was disabled or errored — do NOT conclude the
671
- paper is unpublished).
713
+ Returns (match, status):
714
+ "found" — verified publication match
715
+ "not_found" — EVERY source answered cleanly with no hit; trustworthy
716
+ "incomplete" — some sources answered (no hit) but others were
717
+ disabled/erroring; a batch run that tripped DBLP's rate
718
+ limit lands here — do NOT conclude "unpublished"
719
+ "unavailable" — no source answered at all
672
720
  """
673
721
  from . import cache
674
722
 
@@ -679,6 +727,8 @@ def find_published(
679
727
  return Match(**cached), "found"
680
728
 
681
729
  clean_misses = 0
730
+ # Core sources lost earlier in this run taint this query's verdict too.
731
+ incomplete = any(n in CORE_SOURCES for n in _DISABLED)
682
732
  for name, fn in CASCADE:
683
733
  if name in _DISABLED:
684
734
  continue
@@ -691,8 +741,10 @@ def find_published(
691
741
  _log(f"[{name}] no publication found")
692
742
  except SourceUnavailable as e:
693
743
  _DISABLED[name] = str(e)
744
+ incomplete = incomplete or name in CORE_SOURCES
694
745
  _log(f"[{name}] disabled for the rest of this run: {e}")
695
746
  except Exception as e: # network hiccup on one source must not kill the run
747
+ incomplete = incomplete or name in CORE_SOURCES
696
748
  _log(f"[{name}] error: {type(e).__name__}: {e}")
697
749
 
698
750
  # Exact-title search missed everywhere. Before concluding "no published
@@ -707,6 +759,10 @@ def find_published(
707
759
  clean_misses += 1
708
760
  except SourceUnavailable as e:
709
761
  _DISABLED["dblp"] = str(e)
762
+ incomplete = True
710
763
  except Exception as e:
764
+ incomplete = True
711
765
  _log(f"[dblp-fuzzy] error: {type(e).__name__}: {e}")
712
- return None, ("not_found" if clean_misses else "unavailable")
766
+ if not clean_misses:
767
+ return None, "unavailable"
768
+ return None, ("incomplete" if incomplete else "not_found")
@@ -0,0 +1,86 @@
1
+ """Regression tests for the third round of field-use reports."""
2
+
3
+ from pathlib import Path
4
+
5
+ from bibcite.bibfile import (
6
+ _scrub_month_strings,
7
+ find_existing,
8
+ load_bib_file,
9
+ upsert_entry,
10
+ )
11
+ from bibcite.normalize import fix_pages
12
+
13
+
14
+ def test_fix_pages_dashes():
15
+ assert fix_pages("411–430") == "411--430" # en-dash
16
+ assert fix_pages("411-430") == "411--430" # single hyphen
17
+ assert fix_pages("411 -- 430") == "411--430"
18
+ assert fix_pages("723—726") == "723--726" # em-dash
19
+ assert fix_pages("e123") == "e123" # no range untouched
20
+
21
+
22
+ PREPRINT = {
23
+ "ENTRYTYPE": "misc",
24
+ "ID": "old",
25
+ "title": "An Information-Theoretic Perspective on VICReg",
26
+ "author": "Ravid Shwartz-Ziv and Yann LeCun",
27
+ "howpublished": "arXiv preprint arXiv:2303.00633",
28
+ "eprint": "2303.00633",
29
+ "year": "2023",
30
+ }
31
+
32
+
33
+ def test_find_existing_by_doi_in_url(tmp_path: Path):
34
+ bib = tmp_path / "d.bib"
35
+ upsert_entry(
36
+ bib,
37
+ {
38
+ "ENTRYTYPE": "article",
39
+ "ID": "k",
40
+ "title": "T",
41
+ "author": "A B",
42
+ "journal": "J",
43
+ "year": "2000",
44
+ "url": "https://doi.org/10.1093/biomet/70.3.723",
45
+ },
46
+ )
47
+ db = load_bib_file(bib)
48
+ # No doi field on the entry — matched via the url (pre-0.4.0 files).
49
+ assert find_existing(db, "", doi="10.1093/biomet/70.3.723") is not None
50
+
51
+
52
+ def test_dedupe_catches_title_drift_pair(tmp_path: Path):
53
+ bib = tmp_path / "p.bib"
54
+ upsert_entry(bib, dict(PREPRINT))
55
+ published = {
56
+ "ENTRYTYPE": "inproceedings",
57
+ "ID": "new",
58
+ "title": "An Information Theory Perspective on VICReg", # drifted
59
+ "author": "Ravid Shwartz-Ziv and Yann LeCun",
60
+ "booktitle": "Advances in Neural Information Processing Systems (NeurIPS)",
61
+ "year": "2023",
62
+ }
63
+ action, key = upsert_entry(bib, published)
64
+ # Fuzzy same-author dedupe: upgraded in place, NOT added as a duplicate.
65
+ assert (action, key) == ("upgraded", "old")
66
+ assert bib.read_text().count("@") == 1
67
+
68
+
69
+ def test_scrub_orphan_month_strings(tmp_path: Path):
70
+ bib = tmp_path / "m.bib"
71
+ bib.write_text(
72
+ "@string{january = {January}}\n@string{june = {June}}\n"
73
+ "@article{x, title = {T}, author = {A B}, year = {2000} }\n"
74
+ )
75
+ _scrub_month_strings(bib)
76
+ text = bib.read_text()
77
+ assert "@string" not in text
78
+ assert "title" in text
79
+
80
+
81
+ def test_scrub_leaves_clean_files_alone(tmp_path: Path):
82
+ bib = tmp_path / "c.bib"
83
+ original = "@article{x,\n title = {T},\n author = {A B},\n year = {2000},\n}\n"
84
+ bib.write_text(original)
85
+ _scrub_month_strings(bib)
86
+ assert bib.read_text() == original # untouched, not even rewritten
@@ -0,0 +1,78 @@
1
+ """The batch-429 poisoning scenario: a disabled core source must taint the
2
+ verdict ('incomplete'), never masquerade as a trustworthy 'not_found'."""
3
+
4
+ import pytest
5
+
6
+ import bibcite.sources as sources
7
+ from bibcite import cache
8
+ from bibcite.sources import SourceUnavailable, find_published
9
+
10
+
11
+ @pytest.fixture(autouse=True)
12
+ def isolated(monkeypatch, tmp_path):
13
+ monkeypatch.setattr(cache, "DISABLED", True)
14
+ monkeypatch.setattr(sources, "_DISABLED", {})
15
+ # No fuzzy fallback network calls in these tests.
16
+ monkeypatch.setattr(sources, "try_dblp_fuzzy", lambda *a, **k: None)
17
+
18
+
19
+ def _cascade(**outcomes):
20
+ """Build a fake CASCADE; outcome per source: None=clean miss, 'raise'=429."""
21
+
22
+ def make(o):
23
+ def fn(t, y, a, au):
24
+ if o == "raise":
25
+ raise SourceUnavailable("simulated 429")
26
+ return o
27
+
28
+ return fn
29
+
30
+ return tuple((name, make(o)) for name, o in outcomes.items())
31
+
32
+
33
+ def test_all_clean_misses_is_trustworthy(monkeypatch):
34
+ monkeypatch.setattr(
35
+ sources, "CASCADE", _cascade(dblp=None, semanticscholar=None, crossref=None)
36
+ )
37
+ match, status = find_published("Some Title", author_hint="smith")
38
+ assert (match, status) == (None, "not_found")
39
+
40
+
41
+ def test_core_source_429_taints_verdict(monkeypatch):
42
+ # DBLP 429s, others answer cleanly — the user's exact failure chain.
43
+ monkeypatch.setattr(
44
+ sources,
45
+ "CASCADE",
46
+ _cascade(dblp="raise", googlescholar=None, crossref=None),
47
+ )
48
+ match, status = find_published("Some Title", author_hint="smith")
49
+ assert (match, status) == (None, "incomplete")
50
+
51
+
52
+ def test_previously_disabled_core_source_taints_next_queries(monkeypatch):
53
+ # Query N tripped DBLP; queries N+1... in the same run inherit the taint.
54
+ monkeypatch.setattr(sources, "_DISABLED", {"dblp": "429"})
55
+ monkeypatch.setattr(
56
+ sources, "CASCADE", _cascade(dblp=None, crossref=None) # dblp skipped anyway
57
+ )
58
+ match, status = find_published("Another Title", author_hint="smith")
59
+ assert (match, status) == (None, "incomplete")
60
+
61
+
62
+ def test_noncore_outage_does_not_taint(monkeypatch):
63
+ # Google Scholar captcha is routine; a miss stays trustworthy.
64
+ monkeypatch.setattr(
65
+ sources,
66
+ "CASCADE",
67
+ _cascade(dblp=None, googlescholar="raise", crossref=None),
68
+ )
69
+ match, status = find_published("Some Title", author_hint="smith")
70
+ assert (match, status) == (None, "not_found")
71
+
72
+
73
+ def test_everything_down_is_unavailable(monkeypatch):
74
+ monkeypatch.setattr(
75
+ sources, "CASCADE", _cascade(dblp="raise", semanticscholar="raise")
76
+ )
77
+ match, status = find_published("Some Title", author_hint="smith")
78
+ assert (match, status) == (None, "unavailable")
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.4.0"
21
+ version = "0.5.0"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes
File without changes