bibcite-cli 0.4.0__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/PKG-INFO +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/pyproject.toml +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/__init__.py +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/bibfile.py +51 -4
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/cli.py +51 -13
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/normalize.py +18 -4
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/resolve.py +9 -1
- bibcite_cli-0.4.1/tests/test_round3.py +86 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/uv.lock +1 -1
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/.gitignore +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/LICENSE +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/Readme.md +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/cache.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/data/strings.bib +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/sources.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/src/bibcite/venues.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_bibfile.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_bugfixes.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_entry_types.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_normalize.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_round2.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_strings_override.py +0 -0
- {bibcite_cli-0.4.0 → bibcite_cli-0.4.1}/tests/test_venues.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bibcite-cli
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: Resolve papers (arXiv id / DOI / title) to canonical, normalized BibTeX for agents and humans
|
|
5
5
|
Project-URL: Repository, https://github.com/leo1oel/bibcite
|
|
6
6
|
License-Expression: MIT
|
|
@@ -133,15 +133,37 @@ def load_bib_file(path: Path) -> BibDatabase | None:
|
|
|
133
133
|
return None
|
|
134
134
|
|
|
135
135
|
|
|
136
|
-
def find_existing(
|
|
136
|
+
def find_existing(
|
|
137
|
+
db: BibDatabase,
|
|
138
|
+
title: str,
|
|
139
|
+
arxiv_id: str = "",
|
|
140
|
+
doi: str = "",
|
|
141
|
+
author: str = "",
|
|
142
|
+
) -> dict | None:
|
|
143
|
+
from .normalize import first_author_last_name, titles_similar
|
|
144
|
+
|
|
137
145
|
ref = norm_title(title)
|
|
138
146
|
for entry in db.entries:
|
|
139
147
|
if arxiv_id and entry_arxiv_id(entry) == arxiv_id:
|
|
140
148
|
return entry
|
|
141
|
-
if doi
|
|
142
|
-
|
|
149
|
+
if doi:
|
|
150
|
+
d = doi.lower()
|
|
151
|
+
# Older entries may lack a doi field but carry it in the url.
|
|
152
|
+
if entry.get("doi", "").lower() == d or d in entry.get("url", "").lower():
|
|
153
|
+
return entry
|
|
143
154
|
if ref and norm_title(entry.get("title", "")) == ref:
|
|
144
155
|
return entry
|
|
156
|
+
# Fuzzy pass: title drift (arXiv vs camera-ready) with the same first
|
|
157
|
+
# author is the same paper — catch it BEFORE writing a duplicate pair.
|
|
158
|
+
if title and author:
|
|
159
|
+
last = first_author_last_name(author)
|
|
160
|
+
for entry in db.entries:
|
|
161
|
+
if not entry.get("author"):
|
|
162
|
+
continue
|
|
163
|
+
if first_author_last_name(entry["author"]) != last:
|
|
164
|
+
continue
|
|
165
|
+
if titles_similar(title, entry.get("title", "")):
|
|
166
|
+
return entry
|
|
145
167
|
return None
|
|
146
168
|
|
|
147
169
|
|
|
@@ -170,7 +192,11 @@ def upsert_entry(
|
|
|
170
192
|
existing = next((e for e in db.entries if e.get("ID") == replace_key), None)
|
|
171
193
|
else:
|
|
172
194
|
existing = find_existing(
|
|
173
|
-
db,
|
|
195
|
+
db,
|
|
196
|
+
entry.get("title", ""),
|
|
197
|
+
entry_arxiv_id(entry),
|
|
198
|
+
entry.get("doi", ""),
|
|
199
|
+
entry.get("author", ""),
|
|
174
200
|
)
|
|
175
201
|
|
|
176
202
|
if existing is not None:
|
|
@@ -232,7 +258,28 @@ def tidy_command() -> list[str] | None:
|
|
|
232
258
|
return None
|
|
233
259
|
|
|
234
260
|
|
|
261
|
+
_MONTH_STRING_BLOCK = re.compile(
|
|
262
|
+
r"@string\s*\{\s*(?:" + "|".join(MONTH_STRINGS) + r")\s*=",
|
|
263
|
+
re.IGNORECASE,
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _scrub_month_strings(path: Path):
|
|
268
|
+
"""Remove orphan month @string blocks left by the pre-0.4 leak.
|
|
269
|
+
bibtex-tidy itself preserves @strings, so tidy alone never cleans them."""
|
|
270
|
+
try:
|
|
271
|
+
if not _MONTH_STRING_BLOCK.search(path.read_text()):
|
|
272
|
+
return
|
|
273
|
+
db = load_bib_file(path)
|
|
274
|
+
if db is not None:
|
|
275
|
+
_write_db(path, db) # _write_db drops the injected month macros
|
|
276
|
+
_log("[bibcite] scrubbed leftover month @string blocks")
|
|
277
|
+
except Exception as e:
|
|
278
|
+
_log(f"[bibcite] month-string scrub skipped: {e}")
|
|
279
|
+
|
|
280
|
+
|
|
235
281
|
def run_tidy(path: Path) -> bool:
|
|
282
|
+
_scrub_month_strings(path)
|
|
236
283
|
cmd = tidy_command()
|
|
237
284
|
if cmd is None:
|
|
238
285
|
_log("[bibcite] bibtex-tidy not found (npm i -g bibtex-tidy); skipping tidy")
|
|
@@ -12,7 +12,7 @@ import time
|
|
|
12
12
|
from pathlib import Path
|
|
13
13
|
|
|
14
14
|
from . import bibfile, cache
|
|
15
|
-
from .normalize import first_author_last_name, norm_title, titles_similar
|
|
15
|
+
from .normalize import first_author_last_name, fix_pages, norm_title, titles_similar
|
|
16
16
|
from .resolve import classify
|
|
17
17
|
from .resolve import (
|
|
18
18
|
NotFound,
|
|
@@ -108,6 +108,8 @@ def _resolve_user_bibtex(text: str) -> Resolved:
|
|
|
108
108
|
entry.pop("journal", None)
|
|
109
109
|
entry["ENTRYTYPE"] = canonical.entry_type
|
|
110
110
|
entry[canonical.bib_field] = canonical.name
|
|
111
|
+
if entry.get("pages"):
|
|
112
|
+
entry["pages"] = fix_pages(entry["pages"])
|
|
111
113
|
published = not bibfile.is_preprint(entry)
|
|
112
114
|
return Resolved(
|
|
113
115
|
entry,
|
|
@@ -117,6 +119,31 @@ def _resolve_user_bibtex(text: str) -> Resolved:
|
|
|
117
119
|
)
|
|
118
120
|
|
|
119
121
|
|
|
122
|
+
def _identity_mismatch(path: Path, target_key: str, new_entry: dict) -> str:
|
|
123
|
+
"""`--key` is a scalpel — warn when the resolved paper does not look like
|
|
124
|
+
the entry it is about to overwrite (no shared arXiv id, DOI, or similar
|
|
125
|
+
title), because the old key would then be citing a different paper."""
|
|
126
|
+
db = bibfile.load_bib_file(path)
|
|
127
|
+
if db is None:
|
|
128
|
+
return ""
|
|
129
|
+
target = next((e for e in db.entries if e.get("ID") == target_key), None)
|
|
130
|
+
if target is None:
|
|
131
|
+
return ""
|
|
132
|
+
old_aid, new_aid = bibfile.entry_arxiv_id(target), bibfile.entry_arxiv_id(new_entry)
|
|
133
|
+
if old_aid and new_aid and old_aid == new_aid:
|
|
134
|
+
return ""
|
|
135
|
+
old_doi, new_doi = target.get("doi", "").lower(), new_entry.get("doi", "").lower()
|
|
136
|
+
if old_doi and new_doi and old_doi == new_doi:
|
|
137
|
+
return ""
|
|
138
|
+
if titles_similar(target.get("title", ""), new_entry.get("title", "")):
|
|
139
|
+
return ""
|
|
140
|
+
return (
|
|
141
|
+
f"replacing '{target_key}' with what looks like a DIFFERENT paper "
|
|
142
|
+
f"('{target.get('title', '')[:50]}' -> '{new_entry.get('title', '')[:50]}'); "
|
|
143
|
+
"the key will no longer describe its contents"
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
120
147
|
def _local_exists(path: Path, query: str) -> str | None:
|
|
121
148
|
"""Local pre-check: if the query is already in the file as a PUBLISHED
|
|
122
149
|
entry, skip the network entirely (makes --from re-runs and repeated adds
|
|
@@ -131,7 +158,12 @@ def _local_exists(path: Path, query: str) -> str | None:
|
|
|
131
158
|
existing = bibfile.find_existing(db, "", doi=value)
|
|
132
159
|
else:
|
|
133
160
|
existing = bibfile.find_existing(db, value)
|
|
134
|
-
if existing is
|
|
161
|
+
if existing is None:
|
|
162
|
+
return None
|
|
163
|
+
if not bibfile.is_preprint(existing):
|
|
164
|
+
return existing["ID"]
|
|
165
|
+
if existing.get("pubstate", "").strip("{}") == "preprint":
|
|
166
|
+
# Confirmed preprint-only: nothing to upgrade, no reason to go online.
|
|
135
167
|
return existing["ID"]
|
|
136
168
|
return None
|
|
137
169
|
|
|
@@ -196,6 +228,11 @@ def cmd_add(args) -> int:
|
|
|
196
228
|
if res is None:
|
|
197
229
|
results.append({"query": query, "action": "failed", "exit_code": code})
|
|
198
230
|
continue
|
|
231
|
+
warning = ""
|
|
232
|
+
if args.key:
|
|
233
|
+
warning = _identity_mismatch(path, args.key, res.entry)
|
|
234
|
+
if warning:
|
|
235
|
+
_log(f"[bibcite] warning: {warning}")
|
|
199
236
|
action, key = bibfile.upsert_entry(
|
|
200
237
|
path, res.entry, replace=args.replace, replace_key=args.key or ""
|
|
201
238
|
)
|
|
@@ -210,17 +247,18 @@ def cmd_add(args) -> int:
|
|
|
210
247
|
results.append({"query": query, "action": action, "exit_code": EXIT_NOT_FOUND})
|
|
211
248
|
continue
|
|
212
249
|
wrote = wrote or action != "exists"
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
250
|
+
result = {
|
|
251
|
+
"query": query,
|
|
252
|
+
"action": action,
|
|
253
|
+
"key": key,
|
|
254
|
+
"title": res.entry.get("title", ""),
|
|
255
|
+
"venue": res.venue or "arXiv (preprint)",
|
|
256
|
+
"published": res.published,
|
|
257
|
+
"source": res.source,
|
|
258
|
+
}
|
|
259
|
+
if warning:
|
|
260
|
+
result["warning"] = warning
|
|
261
|
+
results.append(result)
|
|
224
262
|
|
|
225
263
|
tidied = False
|
|
226
264
|
if wrote and not args.no_tidy:
|
|
@@ -82,14 +82,22 @@ def sig_tokens(title: str) -> set[str]:
|
|
|
82
82
|
return {t for t in tokens if len(t) > 2 and t not in ENGLISH_STOPWORDS}
|
|
83
83
|
|
|
84
84
|
|
|
85
|
-
def titles_similar(a: str, b: str, threshold: float = 0.
|
|
86
|
-
"""Token-
|
|
85
|
+
def titles_similar(a: str, b: str, threshold: float = 0.75) -> bool:
|
|
86
|
+
"""Token-overlap similarity — catches preprint→camera-ready title drift
|
|
87
87
|
("Information-Theoretic Perspective" vs "Information Theory Perspective")
|
|
88
|
-
without matching genuinely different papers.
|
|
88
|
+
without matching genuinely different papers.
|
|
89
|
+
|
|
90
|
+
Uses the overlap coefficient (|∩| / min) rather than Jaccard so one
|
|
91
|
+
changed word in a shortish title still matches; very short titles
|
|
92
|
+
(<=3 significant tokens, e.g. "Deep Learning") must match exactly
|
|
93
|
+
because a single shared word would otherwise dominate."""
|
|
89
94
|
ta, tb = sig_tokens(a), sig_tokens(b)
|
|
90
95
|
if not ta or not tb:
|
|
91
96
|
return False
|
|
92
|
-
|
|
97
|
+
smaller = min(len(ta), len(tb))
|
|
98
|
+
if smaller <= 3:
|
|
99
|
+
return ta == tb
|
|
100
|
+
return len(ta & tb) / smaller >= threshold
|
|
93
101
|
|
|
94
102
|
|
|
95
103
|
def fix_author_caps(author_field: str) -> str:
|
|
@@ -114,6 +122,12 @@ def fix_author_caps(author_field: str) -> str:
|
|
|
114
122
|
return " and ".join(fix_name(n) for n in names)
|
|
115
123
|
|
|
116
124
|
|
|
125
|
+
def fix_pages(pages: str) -> str:
|
|
126
|
+
"""BibTeX page ranges use `--`; CrossRef emits en-dashes (411–430) and
|
|
127
|
+
some sources a single hyphen. Collapse any dash run to `--`."""
|
|
128
|
+
return re.sub(r"\s*[-‐-―]+\s*", "--", pages.strip())
|
|
129
|
+
|
|
130
|
+
|
|
117
131
|
def make_key(author_field: str, year: str | int, title: str) -> str:
|
|
118
132
|
"""Deterministic citation key: <lastname><year><firstword>.
|
|
119
133
|
|
|
@@ -10,7 +10,13 @@ import sys
|
|
|
10
10
|
from dataclasses import dataclass
|
|
11
11
|
|
|
12
12
|
from .bibfile import NOISE_FIELDS, parse_bibtex_entry
|
|
13
|
-
from .normalize import
|
|
13
|
+
from .normalize import (
|
|
14
|
+
clean_title,
|
|
15
|
+
first_author_last_name,
|
|
16
|
+
fix_author_caps,
|
|
17
|
+
fix_pages,
|
|
18
|
+
make_key,
|
|
19
|
+
)
|
|
14
20
|
|
|
15
21
|
|
|
16
22
|
class NotFound(Exception):
|
|
@@ -153,6 +159,8 @@ def _finalize(entry: dict, meta: ArxivMeta | None) -> dict:
|
|
|
153
159
|
# a missing url from the DOI.
|
|
154
160
|
if not url or "dx.doi.org" in url:
|
|
155
161
|
entry["url"] = f"https://doi.org/{entry['doi']}"
|
|
162
|
+
if entry.get("pages"):
|
|
163
|
+
entry["pages"] = fix_pages(entry["pages"])
|
|
156
164
|
author = entry.get("author", "") or "anonymous"
|
|
157
165
|
year = entry.get("year", "") or "XXXX"
|
|
158
166
|
entry["ID"] = make_key(author, year, entry.get("title", ""))
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Regression tests for the third round of field-use reports."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from bibcite.bibfile import (
|
|
6
|
+
_scrub_month_strings,
|
|
7
|
+
find_existing,
|
|
8
|
+
load_bib_file,
|
|
9
|
+
upsert_entry,
|
|
10
|
+
)
|
|
11
|
+
from bibcite.normalize import fix_pages
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_fix_pages_dashes():
|
|
15
|
+
assert fix_pages("411–430") == "411--430" # en-dash
|
|
16
|
+
assert fix_pages("411-430") == "411--430" # single hyphen
|
|
17
|
+
assert fix_pages("411 -- 430") == "411--430"
|
|
18
|
+
assert fix_pages("723—726") == "723--726" # em-dash
|
|
19
|
+
assert fix_pages("e123") == "e123" # no range untouched
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
PREPRINT = {
|
|
23
|
+
"ENTRYTYPE": "misc",
|
|
24
|
+
"ID": "old",
|
|
25
|
+
"title": "An Information-Theoretic Perspective on VICReg",
|
|
26
|
+
"author": "Ravid Shwartz-Ziv and Yann LeCun",
|
|
27
|
+
"howpublished": "arXiv preprint arXiv:2303.00633",
|
|
28
|
+
"eprint": "2303.00633",
|
|
29
|
+
"year": "2023",
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_find_existing_by_doi_in_url(tmp_path: Path):
|
|
34
|
+
bib = tmp_path / "d.bib"
|
|
35
|
+
upsert_entry(
|
|
36
|
+
bib,
|
|
37
|
+
{
|
|
38
|
+
"ENTRYTYPE": "article",
|
|
39
|
+
"ID": "k",
|
|
40
|
+
"title": "T",
|
|
41
|
+
"author": "A B",
|
|
42
|
+
"journal": "J",
|
|
43
|
+
"year": "2000",
|
|
44
|
+
"url": "https://doi.org/10.1093/biomet/70.3.723",
|
|
45
|
+
},
|
|
46
|
+
)
|
|
47
|
+
db = load_bib_file(bib)
|
|
48
|
+
# No doi field on the entry — matched via the url (pre-0.4.0 files).
|
|
49
|
+
assert find_existing(db, "", doi="10.1093/biomet/70.3.723") is not None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_dedupe_catches_title_drift_pair(tmp_path: Path):
|
|
53
|
+
bib = tmp_path / "p.bib"
|
|
54
|
+
upsert_entry(bib, dict(PREPRINT))
|
|
55
|
+
published = {
|
|
56
|
+
"ENTRYTYPE": "inproceedings",
|
|
57
|
+
"ID": "new",
|
|
58
|
+
"title": "An Information Theory Perspective on VICReg", # drifted
|
|
59
|
+
"author": "Ravid Shwartz-Ziv and Yann LeCun",
|
|
60
|
+
"booktitle": "Advances in Neural Information Processing Systems (NeurIPS)",
|
|
61
|
+
"year": "2023",
|
|
62
|
+
}
|
|
63
|
+
action, key = upsert_entry(bib, published)
|
|
64
|
+
# Fuzzy same-author dedupe: upgraded in place, NOT added as a duplicate.
|
|
65
|
+
assert (action, key) == ("upgraded", "old")
|
|
66
|
+
assert bib.read_text().count("@") == 1
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_scrub_orphan_month_strings(tmp_path: Path):
|
|
70
|
+
bib = tmp_path / "m.bib"
|
|
71
|
+
bib.write_text(
|
|
72
|
+
"@string{january = {January}}\n@string{june = {June}}\n"
|
|
73
|
+
"@article{x, title = {T}, author = {A B}, year = {2000} }\n"
|
|
74
|
+
)
|
|
75
|
+
_scrub_month_strings(bib)
|
|
76
|
+
text = bib.read_text()
|
|
77
|
+
assert "@string" not in text
|
|
78
|
+
assert "title" in text
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_scrub_leaves_clean_files_alone(tmp_path: Path):
|
|
82
|
+
bib = tmp_path / "c.bib"
|
|
83
|
+
original = "@article{x,\n title = {T},\n author = {A B},\n year = {2000},\n}\n"
|
|
84
|
+
bib.write_text(original)
|
|
85
|
+
_scrub_month_strings(bib)
|
|
86
|
+
assert bib.read_text() == original # untouched, not even rewritten
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|