bibcite-cli 0.4.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/PKG-INFO +1 -1
  2. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/pyproject.toml +1 -1
  3. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/__init__.py +1 -1
  4. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/bibfile.py +51 -4
  5. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/cli.py +51 -13
  6. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/normalize.py +18 -4
  7. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/resolve.py +9 -1
  8. bibcite_cli-0.4.1/tests/test_round3.py +86 -0
  9. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/uv.lock +1 -1
  10. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/.gitignore +0 -0
  11. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/LICENSE +0 -0
  12. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/Readme.md +0 -0
  13. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/cache.py +0 -0
  14. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/data/strings.bib +0 -0
  15. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/sources.py +0 -0
  16. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/venues.py +0 -0
  17. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_bibfile.py +0 -0
  18. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_bugfixes.py +0 -0
  19. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_entry_types.py +0 -0
  20. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_normalize.py +0 -0
  21. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_round2.py +0 -0
  22. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_strings_override.py +0 -0
  23. {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_venues.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bibcite-cli
3
- Version: 0.4.0
3
+ Version: 0.4.1
4
4
  Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
5
5
  Project-URL: Repository, https://github.com/leo1oel/bibcite
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bibcite-cli"
3
- version = "0.4.0"
3
+ version = "0.4.1"
4
4
  description = "Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans"
5
5
  readme = "Readme.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """bibcite: canonical BibTeX resolution for papers (arXiv id / DOI / title)."""
2
2
 
3
- __version__ = "0.4.0"
3
+ __version__ = "0.4.1"
@@ -133,15 +133,37 @@ def load_bib_file(path: Path) -> BibDatabase | None:
133
133
  return None
134
134
 
135
135
 
136
- def find_existing(db: BibDatabase, title: str, arxiv_id: str = "", doi: str = "") -> dict | None:
136
+ def find_existing(
137
+ db: BibDatabase,
138
+ title: str,
139
+ arxiv_id: str = "",
140
+ doi: str = "",
141
+ author: str = "",
142
+ ) -> dict | None:
143
+ from .normalize import first_author_last_name, titles_similar
144
+
137
145
  ref = norm_title(title)
138
146
  for entry in db.entries:
139
147
  if arxiv_id and entry_arxiv_id(entry) == arxiv_id:
140
148
  return entry
141
- if doi and entry.get("doi", "").lower() == doi.lower():
142
- return entry
149
+ if doi:
150
+ d = doi.lower()
151
+ # Older entries may lack a doi field but carry it in the url.
152
+ if entry.get("doi", "").lower() == d or d in entry.get("url", "").lower():
153
+ return entry
143
154
  if ref and norm_title(entry.get("title", "")) == ref:
144
155
  return entry
156
+ # Fuzzy pass: title drift (arXiv vs camera-ready) with the same first
157
+ # author is the same paper — catch it BEFORE writing a duplicate pair.
158
+ if title and author:
159
+ last = first_author_last_name(author)
160
+ for entry in db.entries:
161
+ if not entry.get("author"):
162
+ continue
163
+ if first_author_last_name(entry["author"]) != last:
164
+ continue
165
+ if titles_similar(title, entry.get("title", "")):
166
+ return entry
145
167
  return None
146
168
 
147
169
 
@@ -170,7 +192,11 @@ def upsert_entry(
170
192
  existing = next((e for e in db.entries if e.get("ID") == replace_key), None)
171
193
  else:
172
194
  existing = find_existing(
173
- db, entry.get("title", ""), entry_arxiv_id(entry), entry.get("doi", "")
195
+ db,
196
+ entry.get("title", ""),
197
+ entry_arxiv_id(entry),
198
+ entry.get("doi", ""),
199
+ entry.get("author", ""),
174
200
  )
175
201
 
176
202
  if existing is not None:
@@ -232,7 +258,28 @@ def tidy_command() -> list[str] | None:
232
258
  return None
233
259
 
234
260
 
261
+ _MONTH_STRING_BLOCK = re.compile(
262
+ r"@string\s*\{\s*(?:" + "|".join(MONTH_STRINGS) + r")\s*=",
263
+ re.IGNORECASE,
264
+ )
265
+
266
+
267
+ def _scrub_month_strings(path: Path):
268
+ """Remove orphan month @string blocks left by the pre-0.4 leak.
269
+ bibtex-tidy itself preserves @strings, so tidy alone never cleans them."""
270
+ try:
271
+ if not _MONTH_STRING_BLOCK.search(path.read_text()):
272
+ return
273
+ db = load_bib_file(path)
274
+ if db is not None:
275
+ _write_db(path, db) # _write_db drops the injected month macros
276
+ _log("[bibcite] scrubbed leftover month @string blocks")
277
+ except Exception as e:
278
+ _log(f"[bibcite] month-string scrub skipped: {e}")
279
+
280
+
235
281
  def run_tidy(path: Path) -> bool:
282
+ _scrub_month_strings(path)
236
283
  cmd = tidy_command()
237
284
  if cmd is None:
238
285
  _log("[bibcite] bibtex-tidy not found (npm i -g bibtex-tidy); skipping tidy")
@@ -12,7 +12,7 @@ import time
12
12
  from pathlib import Path
13
13
 
14
14
  from . import bibfile, cache
15
- from .normalize import first_author_last_name, norm_title, titles_similar
15
+ from .normalize import first_author_last_name, fix_pages, norm_title, titles_similar
16
16
  from .resolve import classify
17
17
  from .resolve import (
18
18
  NotFound,
@@ -108,6 +108,8 @@ def _resolve_user_bibtex(text: str) -> Resolved:
108
108
  entry.pop("journal", None)
109
109
  entry["ENTRYTYPE"] = canonical.entry_type
110
110
  entry[canonical.bib_field] = canonical.name
111
+ if entry.get("pages"):
112
+ entry["pages"] = fix_pages(entry["pages"])
111
113
  published = not bibfile.is_preprint(entry)
112
114
  return Resolved(
113
115
  entry,
@@ -117,6 +119,31 @@ def _resolve_user_bibtex(text: str) -> Resolved:
117
119
  )
118
120
 
119
121
 
122
+ def _identity_mismatch(path: Path, target_key: str, new_entry: dict) -> str:
123
+ """`--key` is a scalpel — warn when the resolved paper does not look like
124
+ the entry it is about to overwrite (no shared arXiv id, DOI, or similar
125
+ title), because the old key would then be citing a different paper."""
126
+ db = bibfile.load_bib_file(path)
127
+ if db is None:
128
+ return ""
129
+ target = next((e for e in db.entries if e.get("ID") == target_key), None)
130
+ if target is None:
131
+ return ""
132
+ old_aid, new_aid = bibfile.entry_arxiv_id(target), bibfile.entry_arxiv_id(new_entry)
133
+ if old_aid and new_aid and old_aid == new_aid:
134
+ return ""
135
+ old_doi, new_doi = target.get("doi", "").lower(), new_entry.get("doi", "").lower()
136
+ if old_doi and new_doi and old_doi == new_doi:
137
+ return ""
138
+ if titles_similar(target.get("title", ""), new_entry.get("title", "")):
139
+ return ""
140
+ return (
141
+ f"replacing '{target_key}' with what looks like a DIFFERENT paper "
142
+ f"('{target.get('title', '')[:50]}' -> '{new_entry.get('title', '')[:50]}'); "
143
+ "the key will no longer describe its contents"
144
+ )
145
+
146
+
120
147
  def _local_exists(path: Path, query: str) -> str | None:
121
148
  """Local pre-check: if the query is already in the file as a PUBLISHED
122
149
  entry, skip the network entirely (makes --from re-runs and repeated adds
@@ -131,7 +158,12 @@ def _local_exists(path: Path, query: str) -> str | None:
131
158
  existing = bibfile.find_existing(db, "", doi=value)
132
159
  else:
133
160
  existing = bibfile.find_existing(db, value)
134
- if existing is not None and not bibfile.is_preprint(existing):
161
+ if existing is None:
162
+ return None
163
+ if not bibfile.is_preprint(existing):
164
+ return existing["ID"]
165
+ if existing.get("pubstate", "").strip("{}") == "preprint":
166
+ # Confirmed preprint-only: nothing to upgrade, no reason to go online.
135
167
  return existing["ID"]
136
168
  return None
137
169
 
@@ -196,6 +228,11 @@ def cmd_add(args) -> int:
196
228
  if res is None:
197
229
  results.append({"query": query, "action": "failed", "exit_code": code})
198
230
  continue
231
+ warning = ""
232
+ if args.key:
233
+ warning = _identity_mismatch(path, args.key, res.entry)
234
+ if warning:
235
+ _log(f"[bibcite] warning: {warning}")
199
236
  action, key = bibfile.upsert_entry(
200
237
  path, res.entry, replace=args.replace, replace_key=args.key or ""
201
238
  )
@@ -210,17 +247,18 @@ def cmd_add(args) -> int:
210
247
  results.append({"query": query, "action": action, "exit_code": EXIT_NOT_FOUND})
211
248
  continue
212
249
  wrote = wrote or action != "exists"
213
- results.append(
214
- {
215
- "query": query,
216
- "action": action,
217
- "key": key,
218
- "title": res.entry.get("title", ""),
219
- "venue": res.venue or "arXiv (preprint)",
220
- "published": res.published,
221
- "source": res.source,
222
- }
223
- )
250
+ result = {
251
+ "query": query,
252
+ "action": action,
253
+ "key": key,
254
+ "title": res.entry.get("title", ""),
255
+ "venue": res.venue or "arXiv (preprint)",
256
+ "published": res.published,
257
+ "source": res.source,
258
+ }
259
+ if warning:
260
+ result["warning"] = warning
261
+ results.append(result)
224
262
 
225
263
  tidied = False
226
264
  if wrote and not args.no_tidy:
@@ -82,14 +82,22 @@ def sig_tokens(title: str) -> set[str]:
82
82
  return {t for t in tokens if len(t) > 2 and t not in ENGLISH_STOPWORDS}
83
83
 
84
84
 
85
- def titles_similar(a: str, b: str, threshold: float = 0.7) -> bool:
86
- """Token-Jaccard similarity — catches preprint→camera-ready title drift
85
+ def titles_similar(a: str, b: str, threshold: float = 0.75) -> bool:
86
+ """Token-overlap similarity — catches preprint→camera-ready title drift
87
87
  ("Information-Theoretic Perspective" vs "Information Theory Perspective")
88
- without matching genuinely different papers."""
88
+ without matching genuinely different papers.
89
+
90
+ Uses the overlap coefficient (|∩| / min) rather than Jaccard so one
91
+ changed word in a shortish title still matches; very short titles
92
+ (<=3 significant tokens, e.g. "Deep Learning") must match exactly
93
+ because a single shared word would otherwise dominate."""
89
94
  ta, tb = sig_tokens(a), sig_tokens(b)
90
95
  if not ta or not tb:
91
96
  return False
92
- return len(ta & tb) / len(ta | tb) >= threshold
97
+ smaller = min(len(ta), len(tb))
98
+ if smaller <= 3:
99
+ return ta == tb
100
+ return len(ta & tb) / smaller >= threshold
93
101
 
94
102
 
95
103
  def fix_author_caps(author_field: str) -> str:
@@ -114,6 +122,12 @@ def fix_author_caps(author_field: str) -> str:
114
122
  return " and ".join(fix_name(n) for n in names)
115
123
 
116
124
 
125
+ def fix_pages(pages: str) -> str:
126
+ """BibTeX page ranges use `--`; CrossRef emits en-dashes (411–430) and
127
+ some sources a single hyphen. Collapse any dash run to `--`."""
128
+ return re.sub(r"\s*[-‐-―]+\s*", "--", pages.strip())
129
+
130
+
117
131
  def make_key(author_field: str, year: str | int, title: str) -> str:
118
132
  """Deterministic citation key: <lastname><year><firstword>.
119
133
 
@@ -10,7 +10,13 @@ import sys
10
10
  from dataclasses import dataclass
11
11
 
12
12
  from .bibfile import NOISE_FIELDS, parse_bibtex_entry
13
- from .normalize import clean_title, first_author_last_name, fix_author_caps, make_key
13
+ from .normalize import (
14
+ clean_title,
15
+ first_author_last_name,
16
+ fix_author_caps,
17
+ fix_pages,
18
+ make_key,
19
+ )
14
20
 
15
21
 
16
22
  class NotFound(Exception):
@@ -153,6 +159,8 @@ def _finalize(entry: dict, meta: ArxivMeta | None) -> dict:
153
159
  # a missing url from the DOI.
154
160
  if not url or "dx.doi.org" in url:
155
161
  entry["url"] = f"https://doi.org/{entry['doi']}"
162
+ if entry.get("pages"):
163
+ entry["pages"] = fix_pages(entry["pages"])
156
164
  author = entry.get("author", "") or "anonymous"
157
165
  year = entry.get("year", "") or "XXXX"
158
166
  entry["ID"] = make_key(author, year, entry.get("title", ""))
@@ -0,0 +1,86 @@
1
+ """Regression tests for the third round of field-use reports."""
2
+
3
+ from pathlib import Path
4
+
5
+ from bibcite.bibfile import (
6
+ _scrub_month_strings,
7
+ find_existing,
8
+ load_bib_file,
9
+ upsert_entry,
10
+ )
11
+ from bibcite.normalize import fix_pages
12
+
13
+
14
+ def test_fix_pages_dashes():
15
+ assert fix_pages("411–430") == "411--430" # en-dash
16
+ assert fix_pages("411-430") == "411--430" # single hyphen
17
+ assert fix_pages("411 -- 430") == "411--430"
18
+ assert fix_pages("723—726") == "723--726" # em-dash
19
+ assert fix_pages("e123") == "e123" # no range untouched
20
+
21
+
22
+ PREPRINT = {
23
+ "ENTRYTYPE": "misc",
24
+ "ID": "old",
25
+ "title": "An Information-Theoretic Perspective on VICReg",
26
+ "author": "Ravid Shwartz-Ziv and Yann LeCun",
27
+ "howpublished": "arXiv preprint arXiv:2303.00633",
28
+ "eprint": "2303.00633",
29
+ "year": "2023",
30
+ }
31
+
32
+
33
+ def test_find_existing_by_doi_in_url(tmp_path: Path):
34
+ bib = tmp_path / "d.bib"
35
+ upsert_entry(
36
+ bib,
37
+ {
38
+ "ENTRYTYPE": "article",
39
+ "ID": "k",
40
+ "title": "T",
41
+ "author": "A B",
42
+ "journal": "J",
43
+ "year": "2000",
44
+ "url": "https://doi.org/10.1093/biomet/70.3.723",
45
+ },
46
+ )
47
+ db = load_bib_file(bib)
48
+ # No doi field on the entry — matched via the url (pre-0.4.0 files).
49
+ assert find_existing(db, "", doi="10.1093/biomet/70.3.723") is not None
50
+
51
+
52
+ def test_dedupe_catches_title_drift_pair(tmp_path: Path):
53
+ bib = tmp_path / "p.bib"
54
+ upsert_entry(bib, dict(PREPRINT))
55
+ published = {
56
+ "ENTRYTYPE": "inproceedings",
57
+ "ID": "new",
58
+ "title": "An Information Theory Perspective on VICReg", # drifted
59
+ "author": "Ravid Shwartz-Ziv and Yann LeCun",
60
+ "booktitle": "Advances in Neural Information Processing Systems (NeurIPS)",
61
+ "year": "2023",
62
+ }
63
+ action, key = upsert_entry(bib, published)
64
+ # Fuzzy same-author dedupe: upgraded in place, NOT added as a duplicate.
65
+ assert (action, key) == ("upgraded", "old")
66
+ assert bib.read_text().count("@") == 1
67
+
68
+
69
+ def test_scrub_orphan_month_strings(tmp_path: Path):
70
+ bib = tmp_path / "m.bib"
71
+ bib.write_text(
72
+ "@string{january = {January}}\n@string{june = {June}}\n"
73
+ "@article{x, title = {T}, author = {A B}, year = {2000} }\n"
74
+ )
75
+ _scrub_month_strings(bib)
76
+ text = bib.read_text()
77
+ assert "@string" not in text
78
+ assert "title" in text
79
+
80
+
81
+ def test_scrub_leaves_clean_files_alone(tmp_path: Path):
82
+ bib = tmp_path / "c.bib"
83
+ original = "@article{x,\n title = {T},\n author = {A B},\n year = {2000},\n}\n"
84
+ bib.write_text(original)
85
+ _scrub_month_strings(bib)
86
+ assert bib.read_text() == original # untouched, not even rewritten
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "bibcite-cli"
21
- version = "0.4.0"
21
+ version = "0.4.1"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "bibtexparser" },
File without changes
File without changes
File without changes