scholarcheck 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,50 @@
1
+ """scholarcheck - verifiable literature grounding from the command line.
2
+
3
+ Never cite a paper that does not exist. Every answer is backed by live
4
+ metadata from OpenAlex, Semantic Scholar, Crossref and arXiv.
5
+
6
+ Command line::
7
+
8
+ scholarcheck verify "Attention Is All You Need"
9
+ scholarcheck bibtex "10.1038/s41586-025-10014-0"
10
+ scholarcheck priorart "conformal risk control" -n 6
11
+
12
+ As a library::
13
+
14
+ from scholarcheck import verify_citation, get_bibtex, search
15
+ paper, confidence = verify_citation("Attention Is All You Need")
16
+ """
17
+
18
+ from .cli import (
19
+ multi_search as search,
20
+ latest,
21
+ bibtex as get_bibtex,
22
+ best_match,
23
+ citedby_openalex as cited_by,
24
+ journal_lookup as journal,
25
+ injournal,
26
+ resolve_pdf,
27
+ download_pdf,
28
+ NET_ERRORS,
29
+ )
30
+
31
+ __version__ = "0.1.0"
32
+
33
+
34
+ def verify_citation(query, n=5):
35
+ """Check whether a citation refers to a real paper.
36
+
37
+ Returns ``(paper, confidence)``, where confidence is the fraction of the
38
+ query's content words covered by the matched title: >=0.75 is a confident
39
+ match, <0.45 means the citation is very likely hallucinated. ``paper`` is
40
+ None when nothing matched.
41
+
42
+ An empty result together with a non-empty :data:`NET_ERRORS` means the
43
+ sources could not be reached - that is *not* evidence the paper is fake.
44
+ """
45
+ return best_match(query, n)
46
+
47
+
48
+ __all__ = ["search", "latest", "get_bibtex", "best_match", "cited_by",
49
+ "journal", "injournal", "resolve_pdf", "download_pdf",
50
+ "verify_citation", "NET_ERRORS", "__version__"]
scholarcheck/_nudge.py ADDED
@@ -0,0 +1,58 @@
1
+ """Mention the repo once or twice, to people who are actually using this.
2
+
3
+ Deliberately quiet: never on the first run, never when stderr is not a
4
+ terminal (so piped and redirected output stays clean), never in CI, and
5
+ never more than twice in the lifetime of an install. `SCHOLARCHECK_NO_NUDGE=1`
6
+ turns it off for good.
7
+ """
8
+ import os
9
+ import sys
10
+ import json
11
+ from pathlib import Path
12
+
13
+ REPO = "GuoCheng24/scholarcheck"
14
+ _SHOW_AT = (5, 25) # run counts at which we say something
15
+ _ENV_OFF = "SCHOLARCHECK_NO_NUDGE"
16
+
17
+
18
+ def _state_path():
19
+ base = os.environ.get("XDG_STATE_HOME") or (Path.home() / ".local" / "state")
20
+ return Path(base) / "scholarcheck" / "usage.json"
21
+
22
+
23
+ def _quiet():
24
+ if os.environ.get(_ENV_OFF):
25
+ return True
26
+ if os.environ.get("CI") or os.environ.get("GITHUB_ACTIONS"):
27
+ return True
28
+ # Not a terminal means someone is piping or redirecting us; stay out of it.
29
+ return not (hasattr(sys.stderr, "isatty") and sys.stderr.isatty())
30
+
31
+
32
+ def record_run():
33
+ """Count this run and, at two points, print a single line to stderr.
34
+
35
+ Any failure here is swallowed: a nudge must never break the tool or
36
+ change its exit status.
37
+ """
38
+ if _quiet():
39
+ return
40
+ try:
41
+ p = _state_path()
42
+ try:
43
+ data = json.loads(p.read_text())
44
+ except Exception:
45
+ data = {}
46
+ n = int(data.get("runs", 0)) + 1
47
+ data["runs"] = n
48
+ p.parent.mkdir(parents=True, exist_ok=True)
49
+ p.write_text(json.dumps(data))
50
+ if n in _SHOW_AT:
51
+ print(
52
+ "\n── scholarcheck has been useful " + str(n) + " times. If it saved you time,\n"
53
+ " a star helps other people find it: https://github.com/" + REPO + "\n"
54
+ " (silence this with SCHOLARCHECK_NO_NUDGE=1)",
55
+ file=sys.stderr,
56
+ )
57
+ except Exception:
58
+ pass
scholarcheck/cli.py ADDED
@@ -0,0 +1,803 @@
1
+ #!/usr/bin/env python3
2
+ """scholarcheck - verifiable literature grounding from the command line.
3
+
4
+ Built for one job: **never cite a paper that does not exist.** Every result
5
+ comes back with real metadata (DOI, authors, year, venue) pulled live from
6
+ public scholarly databases, so a reference can be checked rather than trusted.
7
+
8
+ Sources
9
+ OpenAlex primary, stable, no API key required
10
+ Semantic Scholar citations and TLDRs; often rate-limits without a key
11
+ (set SCHOLARCHECK_S2KEY, free to obtain)
12
+ Crossref DOI -> BibTeX
13
+ arXiv preprints, including theory papers OpenAlex has not indexed
14
+
15
+ Commands
16
+ verify is this citation real? -> match confidence, or "likely hallucinated"
17
+ bibtex DOI / title -> a BibTeX entry (refuses to guess on a weak match)
18
+ search multi-source search, re-ranked by term overlap
19
+ latest recent work only - relevance *and* recency, so new papers are not
20
+ buried under highly-cited old ones
21
+ priorart pull the nearest N real papers for a specific claim, with a
22
+ checklist for judging whether the claim is already taken
23
+ citedby what cited a given paper (has someone already extended it?)
24
+ journal live journal metrics, instead of quoting an impact factor from memory
25
+ injournal recent papers from one journal, to study its actual conventions
26
+ fetch download the open-access PDF so a claim can be checked in full text
27
+
28
+ Environment
29
+ SCHOLARCHECK_PROXY optional proxy, e.g. socks5h://127.0.0.1:1080 (default: direct)
30
+ SCHOLARCHECK_MAILTO your email; joins OpenAlex's polite pool for better rate limits
31
+ SCHOLARCHECK_S2KEY optional Semantic Scholar API key
32
+
33
+ Notes learned the hard way
34
+ * Feed **focused keywords**, not a whole sentence - long claims drag in
35
+ off-topic papers.
36
+ * `search` ranks by relevance and therefore favours highly-cited older work.
37
+ To see what is happening *now*, use `latest`.
38
+ * A weak title match returns nothing rather than a plausible-looking wrong
39
+ entry. Silently citing the wrong paper is worse than citing none.
40
+ """
41
+ import os, json, subprocess, argparse, urllib.parse, re, time, datetime
42
+
43
+ CUR_YEAR = datetime.date.today().year # resolved at runtime, never hard-coded
44
+ PROXY = os.environ.get("SCHOLARCHECK_PROXY", "") # direct connection by default
45
+ MAILTO = os.environ.get("SCHOLARCHECK_MAILTO", "") # set to join OpenAlex polite pool
46
+ S2KEY = os.environ.get("SCHOLARCHECK_S2KEY")
47
+ DOI_RE = r"10\.\d{4,9}/[-._;()/:A-Za-z0-9]+"
48
+ ARXIV_RE = r"\d{4}\.\d{4,5}(v\d+)?"
49
+
50
+ #: Network failures seen during this run. This matters: without it, a dead
51
+ #: connection looks exactly like "no such paper exists", and a citation
52
+ #: verifier that reports a network outage as "hallucinated" is worse than
53
+ #: useless. Callers check this before concluding anything is fake.
54
+ NET_ERRORS = []
55
+
56
+
57
+ def _clean_env():
58
+ """Environment for curl, with every inherited *_proxy variable stripped.
59
+
60
+ Proxy behaviour must be decided solely by SCHOLARCHECK_PROXY. Inheriting
61
+ the caller's http_proxy/all_proxy makes the tool behave differently on two
62
+ machines for no visible reason, and mixing an inherited HTTP proxy with an
63
+ explicit SOCKS one fails in ways that are painful to debug.
64
+ """
65
+ return {k: v for k, v in os.environ.items() if "proxy" not in k.lower()}
66
+
67
+
68
+ def _curl(url, accept=None, headers=None, retries=3, timeout=18):
69
+ """Fetch via curl with backoff on 429/503. Returns the body, or None.
70
+
71
+ On failure the reason is recorded in NET_ERRORS so the caller can tell a
72
+ genuine "not found" apart from "could not reach the API".
73
+ """
74
+ cmd = ["curl", "-sS", "-L", "--max-time", str(timeout), "-w", "\n__HTTP__%{http_code}"]
75
+ if PROXY:
76
+ cmd += ["-x", PROXY]
77
+ if accept:
78
+ cmd += ["-H", f"Accept: {accept}"]
79
+ for k, v in (headers or {}).items():
80
+ cmd += ["-H", f"{k}: {v}"]
81
+ ua = "scholarcheck/0.1 (+https://github.com/GuoCheng24/scholarcheck)"
82
+ if MAILTO:
83
+ ua += " mailto:%s" % MAILTO
84
+ cmd += ["-H", "User-Agent: " + ua, url]
85
+ host = urllib.parse.urlsplit(url).netloc
86
+ last = "unknown error"
87
+ for k in range(retries):
88
+ try:
89
+ r = subprocess.run(cmd, capture_output=True, text=True,
90
+ env=_clean_env(), timeout=timeout + 6)
91
+ out, code = r.stdout, ""
92
+ if "__HTTP__" in out:
93
+ out, _, code = out.rpartition("__HTTP__"); code = code.strip()
94
+ if r.returncode == 0 and out.strip() and code in ("200", "201", ""):
95
+ return out
96
+ if r.returncode != 0:
97
+ last = (r.stderr or "").strip().splitlines()[-1] if r.stderr else f"curl exit {r.returncode}"
98
+ elif code:
99
+ last = f"HTTP {code}"
100
+ if code in ("429", "403", "503", "500", "502"): # rate-limited or temporarily unavailable -> back off
101
+ time.sleep(2.0 * (k + 1)); continue
102
+ except Exception as e:
103
+ last = f"{type(e).__name__}: {e}"
104
+ time.sleep(1.0 * (k + 1))
105
+ NET_ERRORS.append(f"{host}: {last}")
106
+ return None
107
+
108
+
109
+ #: Sources whose failure genuinely means "we could not look". Semantic Scholar
110
+ #: rate-limits hard without an API key and arXiv only supplements coverage, so
111
+ #: neither should turn a real answer into "inconclusive" - otherwise the tool
112
+ #: would refuse to flag anything whenever S2 returns 429, which is often.
113
+ _PRIMARY = ("openalex", "crossref")
114
+
115
+
116
+ def _net_failed(primary_only=True):
117
+ """True if a source we actually depend on could not be reached."""
118
+ if primary_only:
119
+ return any(any(h in e for h in _PRIMARY) for e in NET_ERRORS)
120
+ return bool(NET_ERRORS)
121
+
122
+
123
+ def _net_hint():
124
+ """A one-line, actionable explanation of what went wrong on the network."""
125
+ crit = [e for e in NET_ERRORS if any(h in e for h in _PRIMARY)]
126
+ uniq = list(dict.fromkeys(crit or NET_ERRORS))[:3]
127
+ hint = "Could not reach: " + "; ".join(uniq)
128
+ if PROXY:
129
+ hint += f"\n (using proxy {PROXY} from SCHOLARCHECK_PROXY - check it is reachable)"
130
+ else:
131
+ hint += "\n (no proxy set; if your network needs one, set SCHOLARCHECK_PROXY)"
132
+ return hint
133
+
134
+ def _get_json(url, headers=None):
135
+ t = _curl(url, accept="application/json", headers=headers)
136
+ if not t:
137
+ return None
138
+ try:
139
+ return json.loads(t)
140
+ except Exception:
141
+ return None
142
+
143
+ # ---------- text utilities ----------
144
+ _STOP = set("the a an of for and or to in on with via using from is are be that this we our by as at "
145
+ "how when what which under over into onto not no".split())
146
+ def _tokens(s, minlen=4):
147
+ return [w for w in re.findall(r"[a-zA-Z][a-zA-Z\-]+", (s or "").lower()) if len(w) >= minlen and w not in _STOP]
148
+ def _match_ratio(query, title):
149
+ """Fraction of the query's content words covered by the title (used by `verify`)."""
150
+ tq = set(_tokens(query, 3)); tt = set(_tokens(title, 3))
151
+ if not tq or not tt:
152
+ return 0.0
153
+ return len(tq & tt) / len(tq)
154
+ def _relevance(query, p):
155
+ """Term hits weighted title x3 + abstract x1; used to re-rank away off-topic results."""
156
+ toks = set(_tokens(query))
157
+ title = (p.get("title") or "").lower(); ab = (p.get("abstract") or "").lower()
158
+ return sum((3 if w in title else 0) + (1 if w in ab else 0) for w in toks)
159
+
160
+ # ---------- OpenAlex ----------
161
+ def _oa_abstract(inv):
162
+ if not inv:
163
+ return ""
164
+ pos = {}
165
+ for w, idxs in inv.items():
166
+ for i in idxs:
167
+ pos[i] = w
168
+ s = " ".join(pos[i] for i in sorted(pos))
169
+ return (s[:300] + "…") if len(s) > 300 else s
170
+
171
+ def _oa_work(w):
172
+ src = (w.get("primary_location") or {}).get("source") or {}
173
+ ids = w.get("ids") or {}
174
+ return {
175
+ "title": w.get("title") or "(no title)",
176
+ "year": w.get("publication_year"),
177
+ "venue": src.get("display_name") or (w.get("type") or ""),
178
+ "doi": (w.get("doi") or "").replace("https://doi.org/", "") or None,
179
+ "arxiv": None,
180
+ "id": (w.get("id") or "").replace("https://openalex.org/", ""),
181
+ "cited_by": w.get("cited_by_count", 0),
182
+ "authors": [a.get("author", {}).get("display_name") for a in (w.get("authorships") or [])[:6]],
183
+ "abstract": _oa_abstract(w.get("abstract_inverted_index")),
184
+ }
185
+
186
+ def search_openalex(query, n, since=None):
187
+ q = urllib.parse.quote(query)
188
+ url = f"https://api.openalex.org/works?search={q}&per-page={n}&sort=relevance_score:desc&mailto={MAILTO}"
189
+ if since:
190
+ url += f"&filter=from_publication_date:{since}-01-01"
191
+ d = _get_json(url)
192
+ if not d or "results" not in d:
193
+ return None
194
+ return [_oa_work(w) for w in d["results"]]
195
+
196
+ def citedby_openalex(oa_id, n):
197
+ url = f"https://api.openalex.org/works?filter=cites:{oa_id}&per-page={n}&sort=cited_by_count:desc&mailto={MAILTO}"
198
+ d = _get_json(url)
199
+ if not d or "results" not in d:
200
+ return None
201
+ return [_oa_work(w) for w in d["results"]]
202
+
203
+ def journal_lookup(name, n=5):
204
+ """Live journal metrics. OpenAlex 2yr_mean_citedness is an IF-like measure:
205
+ open, current, and not behind the JCR paywall - but not the official IF."""
206
+ q = urllib.parse.quote(name)
207
+ d = _get_json(f"https://api.openalex.org/sources?search={q}&per-page={n}&mailto={MAILTO}")
208
+ if not d or "results" not in d:
209
+ return None
210
+ out = []
211
+ for s in d["results"]:
212
+ ss = s.get("summary_stats") or {}
213
+ out.append({
214
+ "name": s.get("display_name"), "type": s.get("type"),
215
+ "id": (s.get("id") or "").replace("https://openalex.org/", ""),
216
+ "if2yr": ss.get("2yr_mean_citedness"), "h_index": ss.get("h_index"),
217
+ "works": s.get("works_count"), "issn": s.get("issn_l"),
218
+ "publisher": s.get("host_organization_name"), "homepage": s.get("homepage_url"),
219
+ })
220
+ return out
221
+
222
+ def injournal(name, n, topic=None, since=None):
223
+ """Recent papers from one journal, to study its actual conventions. Returns (name, papers)."""
224
+ js = journal_lookup(name, 1)
225
+ if not js or not js[0].get("id"):
226
+ return None, None
227
+ sid = js[0]["id"]; jname = js[0]["name"]
228
+ yr = int(since) if since else CUR_YEAR - 2
229
+ url = (f"https://api.openalex.org/works?filter=primary_location.source.id:{sid},"
230
+ f"from_publication_date:{yr}-01-01&per-page={n * 3}&sort=publication_date:desc&mailto={MAILTO}")
231
+ if topic:
232
+ url += f"&search={urllib.parse.quote(topic)}"
233
+ d = _get_json(url)
234
+ if not d or "results" not in d:
235
+ return jname, None
236
+ ws = []
237
+ for w in d["results"]:
238
+ p = _oa_work(w)
239
+ p["is_oa"] = bool((w.get("open_access") or {}).get("is_oa"))
240
+ ws.append(p)
241
+ if topic:
242
+ ws.sort(key=lambda p: (_relevance(topic, p), p.get("year") or 0), reverse=True)
243
+ return jname, ws[:n]
244
+
245
+ def resolve_openalex_id(s):
246
+ """Title / DOI / OpenAlex id -> an OpenAlex work id (used by `citedby`)."""
247
+ s = s.strip()
248
+ if re.fullmatch(r"W\d+", s):
249
+ return s
250
+ m = re.search(DOI_RE, s)
251
+ if m:
252
+ d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(m.group(0))}&per-page=1&mailto={MAILTO}")
253
+ if d and d.get("results"):
254
+ return d["results"][0]["id"].replace("https://openalex.org/", "")
255
+ best, r = best_match(s) # titles go through best_match; too weak -> refuse to resolve
256
+ if not best or r < 0.5:
257
+ return None
258
+ if (best.get("id") or "").startswith("W"):
259
+ return best["id"]
260
+ if best.get("doi"): # matched in S2/arXiv -> exchange the DOI for an OpenAlex id
261
+ d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(best['doi'])}&per-page=1&mailto={MAILTO}")
262
+ if d and d.get("results"):
263
+ return d["results"][0]["id"].replace("https://openalex.org/", "")
264
+ return None
265
+
266
+ # ---------- Semantic Scholar ----------
267
+ def search_s2(query, n):
268
+ q = urllib.parse.quote(query)
269
+ fields = "title,year,venue,authors,externalIds,abstract,citationCount,tldr"
270
+ url = f"https://api.semanticscholar.org/graph/v1/paper/search?query={q}&limit={n}&fields={fields}"
271
+ d = _get_json(url, headers=({"x-api-key": S2KEY} if S2KEY else None))
272
+ if not d or "data" not in d:
273
+ return None
274
+ out = []
275
+ for p in d["data"]:
276
+ ext = p.get("externalIds") or {}
277
+ out.append({
278
+ "title": p.get("title"), "year": p.get("year"),
279
+ "venue": p.get("venue") or "", "doi": ext.get("DOI"),
280
+ "arxiv": ext.get("ArXiv"), "id": p.get("paperId"),
281
+ "cited_by": p.get("citationCount", 0),
282
+ "authors": [a.get("name") for a in (p.get("authors") or [])[:6]],
283
+ "abstract": (p.get("tldr") or {}).get("text") or (p.get("abstract") or "")[:300],
284
+ })
285
+ return out
286
+
287
+ # ---------- arXiv ----------
288
+ def search_arxiv(query, n):
289
+ import xml.etree.ElementTree as ET
290
+ q = urllib.parse.quote(query)
291
+ url = f"https://export.arxiv.org/api/query?search_query=all:{q}&start=0&max_results={n}&sortBy=relevance"
292
+ t = _curl(url)
293
+ if not t:
294
+ return None
295
+ try:
296
+ root = ET.fromstring(t)
297
+ except Exception:
298
+ return None
299
+ ns = {"a": "http://www.w3.org/2005/Atom"}
300
+ out = []
301
+ for e in root.findall("a:entry", ns):
302
+ aid = (e.findtext("a:id", default="", namespaces=ns) or "").split("/abs/")[-1]
303
+ yr = (e.findtext("a:published", default="", namespaces=ns) or "")[:4]
304
+ out.append({
305
+ "title": " ".join((e.findtext("a:title", default="", namespaces=ns) or "").split()),
306
+ "year": int(yr) if yr.isdigit() else None,
307
+ "venue": "arXiv", "doi": None, "arxiv": aid, "id": aid, "cited_by": 0,
308
+ "authors": [a.findtext("a:name", default="", namespaces=ns) for a in e.findall("a:author", ns)][:6],
309
+ "abstract": " ".join((e.findtext("a:summary", default="", namespaces=ns) or "").split())[:300],
310
+ })
311
+ return out
312
+
313
+ def arxiv_by_id(aid):
314
+ """Look an arXiv id up on arXiv itself. Returns one paper dict, or None.
315
+
316
+ This is the authoritative mapping for an arXiv id, and it is used in
317
+ preference to resolving the id through a DOI. Aggregators derive the
318
+ 10.48550/arXiv.* DOI second-hand and can attach it to the wrong record -
319
+ observed in the wild, returning an unrelated paper for a valid id, which is
320
+ far more damaging than returning nothing.
321
+ """
322
+ import xml.etree.ElementTree as ET
323
+
324
+ aid = aid.split("v")[0]
325
+ t = _curl(f"https://export.arxiv.org/api/query?id_list={urllib.parse.quote(aid)}")
326
+ if not t:
327
+ return None
328
+ try:
329
+ root = ET.fromstring(t)
330
+ except Exception:
331
+ return None
332
+ ns = {"a": "http://www.w3.org/2005/Atom"}
333
+ e = root.find("a:entry", ns)
334
+ if e is None:
335
+ return None
336
+ title = " ".join((e.findtext("a:title", default="", namespaces=ns) or "").split())
337
+ if not title or title.lower().startswith("error"):
338
+ return None
339
+ yr = (e.findtext("a:published", default="", namespaces=ns) or "")[:4]
340
+ doi = e.findtext("a:doi", default="", namespaces=ns) or None
341
+ return {
342
+ "title": title,
343
+ "year": int(yr) if yr.isdigit() else None,
344
+ "venue": "arXiv", "doi": doi, "arxiv": aid, "id": aid, "cited_by": 0,
345
+ "authors": [a.findtext("a:name", default="", namespaces=ns)
346
+ for a in e.findall("a:author", ns)][:6],
347
+ "abstract": " ".join((e.findtext("a:summary", default="", namespaces=ns) or "").split())[:300],
348
+ }
349
+
350
+
351
+ #: Recorded when two sources return materially different records for the same
352
+ #: identifier. Cross-checking is the whole point of querying more than one.
353
+ SOURCE_CONFLICTS = []
354
+
355
+
356
+ # ---------- BibTeX ----------
357
+ def _fallback_bibtex(p):
358
+ """Build an entry from metadata when there is no Crossref DOI (arXiv -> @misc with eprint)."""
359
+ names = [x for x in (p.get("authors") or []) if x]
360
+ auth = " and ".join(names) if names else "Unknown"
361
+ y = p.get("year") or "n.d."
362
+ first = (re.sub(r"[^A-Za-z]", "", (names[0].split()[-1] if names else "anon")) or "anon").lower()
363
+ kw = (_tokens(p.get("title", "")) or ["ref"])[0]
364
+ key = f"{first}{y}{kw}"
365
+ title = p.get("title", "")
366
+ if p.get("arxiv"):
367
+ return (f"@misc{{{key},\n title={{{title}}},\n author={{{auth}}},\n year={{{y}}},\n"
368
+ f" eprint={{{p['arxiv']}}},\n archivePrefix={{arXiv}},\n note={{arXiv:{p['arxiv']}}}\n}}")
369
+ return (f"@article{{{key},\n title={{{title}}},\n author={{{auth}}},\n year={{{y}}},\n"
370
+ f" journal={{{p.get('venue','')}}}\n}}")
371
+
372
+ def bibtex(s):
373
+ """Returns a BibTeX string; ('WEAK', candidate, ratio) when the title match is
374
+ too weak to be safe; ('INCONCLUSIVE', candidate, ratio) when a primary source
375
+ could not be reached; or None. It will not hand back a wrong entry."""
376
+ s = s.strip()
377
+ m = re.search(DOI_RE, s)
378
+ doi = m.group(0) if m else None
379
+ meta = None
380
+ if not doi: # title input: best_match guards against returning the wrong paper
381
+ NET_ERRORS.clear()
382
+ meta, r = best_match(s)
383
+ # A degraded search is exactly when the wrong entry is most likely: the
384
+ # "best" match is then drawn from whichever sources happened to answer.
385
+ # `verify` already refuses to speak under those conditions; emitting a
386
+ # citation would be a stronger claim than refusing to make a weaker one.
387
+ if _net_failed():
388
+ return ("INCONCLUSIVE", meta, r)
389
+ # Inclusive boundary: half the query's content words is not a confident
390
+ # match by any reading, and an exact 0.50 used to slip through.
391
+ if not meta or r <= 0.5:
392
+ return ("WEAK", meta, r) # better to return nothing than to silently emit a wrong citation
393
+ doi = meta.get("doi")
394
+ if doi:
395
+ bib = _curl(f"https://doi.org/{doi}", accept="application/x-bibtex")
396
+ if bib and "@" in bib:
397
+ return bib.strip()
398
+ if meta is None and doi: # Crossref failed -> fall back to OpenAlex metadata
399
+ d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(doi)}&per-page=1&mailto={MAILTO}")
400
+ if d and d.get("results"):
401
+ meta = _oa_work(d["results"][0])
402
+ return _fallback_bibtex(meta) if meta else None
403
+
404
+ # ---------- PDF retrieval: from a hit to the actual full text ----------
405
+ PDFDIR = os.environ.get("SCHOLAR_PDFDIR", "/tmp/scholar_pdfs")
406
+
407
+ def _openalex_work_full(doi=None, oa_id=None):
408
+ if doi:
409
+ d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(doi)}&per-page=1&mailto={MAILTO}")
410
+ return (d.get("results") or [None])[0] if d else None
411
+ if oa_id:
412
+ return _get_json(f"https://api.openalex.org/works/{oa_id}?mailto={MAILTO}")
413
+ return None
414
+
415
+ def _work_pdf_url(work):
416
+ """Find a downloadable PDF on an OpenAlex work: arXiv first, then OA pdf_url/oa_url."""
417
+ if not work:
418
+ return None, None
419
+ for loc in [work.get("primary_location")] + (work.get("locations") or []):
420
+ if not loc:
421
+ continue
422
+ landing = loc.get("landing_page_url") or ""
423
+ host = ((loc.get("source") or {}).get("display_name") or "")
424
+ if "arxiv" in (landing + host).lower():
425
+ m = re.search(r"(\d{4}\.\d{4,5})", landing)
426
+ if m:
427
+ return f"https://arxiv.org/pdf/{m.group(1)}", "arXiv:" + m.group(1)
428
+ if loc.get("pdf_url"):
429
+ return loc["pdf_url"], "OA-pdf"
430
+ oa = (work.get("open_access") or {}).get("oa_url")
431
+ return (oa, "OA") if oa else (None, None)
432
+
433
+ def resolve_pdf(s):
434
+ """arXiv id / DOI / title / URL -> (pdf_url, label, stem) or (None, reason, None)."""
435
+ s = s.strip()
436
+ m = re.fullmatch(r"(?:arxiv:)?(\d{4}\.\d{4,5})(v\d+)?", s, re.I)
437
+ if m:
438
+ aid = m.group(1) + (m.group(2) or "")
439
+ return f"https://arxiv.org/pdf/{aid}", "arXiv:" + aid, aid.replace(".", "_")
440
+ if s.lower().startswith("http"):
441
+ return s, "url", re.sub(r"\W+", "_", s)[-40:]
442
+ doi_m = re.search(DOI_RE, s)
443
+ work, stem = None, "paper"
444
+ if doi_m:
445
+ work = _openalex_work_full(doi=doi_m.group(0)); stem = doi_m.group(0).replace("/", "_")
446
+ else:
447
+ best, r = best_match(s)
448
+ if not best or r < 0.5:
449
+ return None, "title match too weak - use a DOI, an arXiv id, or a more exact title", None
450
+ stem = re.sub(r"\W+", "_", (best.get("title") or "paper"))[:40]
451
+ if best.get("arxiv"):
452
+ aid = best["arxiv"]
453
+ return f"https://arxiv.org/pdf/{aid}", "arXiv:" + aid, aid.replace(".", "_")
454
+ if (best.get("id") or "").startswith("W"):
455
+ work = _openalex_work_full(oa_id=best["id"])
456
+ elif best.get("doi"):
457
+ work = _openalex_work_full(doi=best["doi"])
458
+ url, label = _work_pdf_url(work)
459
+ return (url, label, stem) if url else (None, "no open-access PDF found (likely paywalled - try the publisher or your library)", None)
460
+
461
+ def download_pdf(url, path, retries=3):
462
+ os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
463
+ env = dict(os.environ); env["no_proxy"] = ""; env["NO_PROXY"] = ""
464
+ cands = [url]
465
+ m = re.search(r"PMC(\d+)", url) # direct PMC links are often blocked -> fall back to the europepmc renderer
466
+ if m:
467
+ cands.append(f"https://europepmc.org/articles/PMC{m.group(1)}?pdf=render")
468
+ UA = "Mozilla/5.0 (X11; Linux x86_64) scholar-ground/1.0"
469
+ for u in cands:
470
+ for k in range(retries): # these endpoints are flaky; retry automatically
471
+ try:
472
+ subprocess.run(["curl", "-sL", "--max-time", "60", "-x", PROXY,
473
+ "-H", f"User-Agent: {UA}", "-o", path, u],
474
+ env=env, timeout=70, capture_output=True)
475
+ if os.path.getsize(path) > 1000 and open(path, "rb").read(5) == b"%PDF-":
476
+ return True
477
+ except Exception:
478
+ pass
479
+ time.sleep(1.2 * (k + 1))
480
+ try: # drop non-PDF junk (error pages) instead of leaving a broken file
481
+ if os.path.exists(path) and open(path, "rb").read(5) != b"%PDF-":
482
+ os.remove(path)
483
+ except Exception:
484
+ pass
485
+ return False
486
+
487
+ # ---------- output ----------
488
+ def _locator(p):
489
+ if p.get("doi"):
490
+ return "doi:" + p["doi"]
491
+ if p.get("arxiv"):
492
+ return "arXiv:" + p["arxiv"]
493
+ pid = p.get("id") or ""
494
+ if pid.startswith("W"):
495
+ return "OpenAlex:" + pid
496
+ return "no locator"
497
+
498
+ def fmt_paper(p, i=None):
499
+ tag = f"[{i}] " if i is not None else ""
500
+ names = [x for x in (p.get("authors") or []) if x]
501
+ au = ", ".join(names[:3]) + (" et al." if len(names) > 3 else "")
502
+ head = f"{tag}{p.get('title') or '(no title)'} ({p.get('year','?')}, {p.get('venue') or '?'}; cited={p.get('cited_by',0)}) {_locator(p)}"
503
+ body = (f" {au}\n {p.get('abstract','')}").rstrip()
504
+ return head + ("\n" + body if body.strip() else "")
505
+
506
+ def multi_search(query, n, since=None):
507
+ """OpenAlex as the stable base, opportunistically enriched with S2 and arXiv,
508
+ de-duplicated, then re-ranked by term overlap."""
509
+ pool, seen = [], set()
510
+ def add(lst):
511
+ for p in (lst or []):
512
+ k = (p.get("title") or "").lower().strip()[:60]
513
+ if k and k not in seen:
514
+ seen.add(k); pool.append(p)
515
+ add(search_openalex(query, n * 3, since))
516
+ add(search_s2(query, n))
517
+ if len(pool) < n:
518
+ add(search_arxiv(query, n))
519
+ if since:
520
+ yr0 = int(since)
521
+ pool = [p for p in pool if (p.get("year") or 0) >= yr0] or pool
522
+ pool.sort(key=lambda p: (_relevance(query, p) + _year_bonus(p), p.get("cited_by", 0)), reverse=True)
523
+ return pool[:n]
524
+
525
+ def _year_bonus(p):
526
+ """Recency weighting, so this year's papers are not buried under highly-cited old ones."""
527
+ y = p.get("year") or 0
528
+ if y >= CUR_YEAR: return 2.5
529
+ if y >= CUR_YEAR - 1: return 1.3
530
+ if y >= CUR_YEAR - 2: return 0.5
531
+ return 0.0
532
+
533
+ def latest(query, n, since=None):
534
+ """Recent related work: relevance ranking plus a recency filter, which avoids the
535
+ noise of sorting by date alone. Defaults to roughly the last 18 months."""
536
+ yr = int(since) if since else CUR_YEAR - 1
537
+ res = (search_openalex(query, n * 2, since=str(yr)) or [])
538
+ s2 = search_s2(query, n) or []
539
+ seen = {(p.get("title") or "").lower()[:60] for p in res}
540
+ for p in s2:
541
+ if (p.get("year") or 0) >= yr and (p.get("title") or "").lower()[:60] not in seen:
542
+ res.append(p)
543
+ res.sort(key=lambda p: (p.get("year") or 0, _relevance(query, p)), reverse=True)
544
+ return res[:n]
545
+
546
+ def resolve_identifier(s):
547
+ """A DOI or arXiv id resolved exactly, or None if `s` is not an identifier.
548
+
549
+ Identifiers must not go through title search. Feeding "arXiv:1906.08253"
550
+ to a title matcher returns whatever paper happens to share those digits and
551
+ then scores it as a mismatch - which reads as "this citation is fake" when
552
+ the truth is that the query was never looked up properly.
553
+ """
554
+ m = re.search(DOI_RE, s)
555
+ if m:
556
+ w = _openalex_work_full(doi=m.group(0))
557
+ return _oa_work(w) if w else None
558
+
559
+ m = re.search(r"(?:arxiv[:\s/]*)?(" + ARXIV_RE + r")", s, re.I)
560
+ if m and (re.search(r"arxiv", s, re.I) or re.fullmatch(r"[\d.v]+", s.strip())):
561
+ aid = m.group(1)
562
+ primary = arxiv_by_id(aid) # authoritative for an arXiv id
563
+ w = _openalex_work_full(doi="10.48550/arXiv." + aid.split("v")[0])
564
+ secondary = _oa_work(w) if w else None
565
+ if primary and secondary:
566
+ # Disagreement means an aggregator has the id attached to the wrong
567
+ # work. Trust arXiv, and say so rather than silently picking one.
568
+ if _match_ratio(primary["title"], secondary["title"]) < 0.5:
569
+ SOURCE_CONFLICTS.append(
570
+ f"arXiv:{aid} -> arXiv says \"{primary['title'][:60]}\"; "
571
+ f"the aggregator says \"{secondary['title'][:60]}\"")
572
+ else:
573
+ primary["cited_by"] = secondary.get("cited_by", 0) or 0
574
+ primary["venue"] = secondary.get("venue") or primary["venue"]
575
+ return primary or secondary
576
+ return None
577
+
578
+
579
+ def best_match(query, n=6):
580
+ """Title -> best matching paper, with re-ranking and a coverage ratio, so bibtex
581
+ and citedby never silently resolve to the wrong paper. Returns (paper|None, ratio)."""
582
+ cands = multi_search(query, n) or []
583
+ best = max(cands, key=lambda p: _match_ratio(query, p.get("title", "")), default=None)
584
+ r = _match_ratio(query, best.get("title", "")) if best else 0.0
585
+ if r < 0.6: # weak match -> also try arXiv directly (theory papers are often arXiv-only)
586
+ for p in (search_arxiv(query, 5) or []):
587
+ rr = _match_ratio(query, p.get("title", ""))
588
+ if rr > r:
589
+ best, r = p, rr
590
+ return best, r
591
+
592
+ OCC_TEMPLATE = (
593
+ "—" * 60 + "\n"
594
+ "[Prior-art check] Ask this of every paper below. All 'no' => the claim is still open:\n"
595
+ " * Does it do *exactly* your specific twist and mechanism? Overlapping on the\n"
596
+ " broader topic alone does not count as taken.\n"
597
+ " * Does it cover the dimension your claim is more specific about - which\n"
598
+ " variable, regime, bound or mechanism?\n"
599
+ " * If one looks close, run `citedby \"<its DOI/title>\"` to see whether a\n"
600
+ " follow-up already covers your extension.\n"
601
+ " ! Never conclude 'already taken' from a title alone. For the closest papers,\n"
602
+ " run `scholarcheck fetch \"<DOI/arXiv-id/title>\"` and check the full text."
603
+ )
604
+
605
+ def _main():
606
+ ap = argparse.ArgumentParser(
607
+ description="Verifiable literature grounding - OpenAlex / Semantic Scholar / Crossref / arXiv",
608
+ epilog='example: scholarcheck priorart "low-degree polynomial detection lower bound" -n 6 --since 2020',
609
+ formatter_class=argparse.RawDescriptionHelpFormatter)
610
+ ap.add_argument("cmd", choices=["search", "priorart", "occupancy", "verify", "bibtex", "citedby", "fetch", "latest", "journal", "injournal"])
611
+ ap.add_argument("query", help="focused keywords / paper title / DOI / arXiv id / URL / journal name")
612
+ ap.add_argument("-n", type=int, default=8, help="number of results (default 8)")
613
+ ap.add_argument("--since", default=None, help="only papers from year YYYY onwards")
614
+ ap.add_argument("--topic", default=None, help="injournal: topic keywords to filter within the journal")
615
+ ap.add_argument("--json", action="store_true", help="structured JSON output")
616
+ ap.add_argument("-o", "--out", default=None, help="fetch: path to save the PDF")
617
+ a = ap.parse_args()
618
+
619
+ if a.cmd in ("search", "priorart", "occupancy"):
620
+ res = multi_search(a.query, a.n, a.since)
621
+ if a.json:
622
+ print(json.dumps(res, ensure_ascii=False, indent=2)); return
623
+ if not res:
624
+ if _net_failed():
625
+ print(f"Could not query the sources.\n {_net_hint()}"); return
626
+ print("No results. Try narrower, more focused keywords - long sentences "
627
+ "drag in off-topic papers."); return
628
+ print(f"# {len(res)} nearest papers - query: {a.query}\n")
629
+ for i, p in enumerate(res, 1):
630
+ print(fmt_paper(p, i)); print()
631
+ if a.cmd in ("priorart", "occupancy"):
632
+ lat = latest(a.query, 5) # always surface the newest work - a prior-art check must not miss it
633
+ if lat:
634
+ print(f"# Most recent related work (>={CUR_YEAR-1}) - check whether someone *just* did your angle:")
635
+ for p in lat: print(" " + fmt_paper(p).replace("\n", "\n "))
636
+ print()
637
+ print(OCC_TEMPLATE)
638
+ return
639
+
640
+ if a.cmd == "injournal":
641
+ jname, res = injournal(a.query, a.n, a.topic, a.since)
642
+ if a.json:
643
+ print(json.dumps(res or [], ensure_ascii=False, indent=2)); return
644
+ if not res:
645
+ print(f"Journal or its recent papers not found: {a.query} (resolved name={jname})"); return
646
+ print(f"# Recent papers in {jname} - study its actual conventions before submitting "
647
+ f"- topic: {a.topic or '(all)'}\n")
648
+ for i, p in enumerate(res, 1):
649
+ oa = "OA" if p.get("is_oa") else "closed"
650
+ print(fmt_paper(p, i) + f" [{oa}]"); print()
651
+ print('-> Use `scholarcheck fetch "<DOI>" -o out.pdf` to pull one, then read it for\n'
652
+ ' structure, figure style, abstract format, length and how statistics are reported.')
653
+ print(' Note: is_oa includes free-to-read (bronze), which is not always machine-downloadable.')
654
+ print(' Some publishers block automated downloads; fall back to an institutional network.')
655
+ return
656
+
657
+ if a.cmd == "journal":
658
+ res = journal_lookup(a.query, a.n)
659
+ if a.json:
660
+ print(json.dumps(res or [], ensure_ascii=False, indent=2)); return
661
+ if not res:
662
+ print(f"Journal not found, or the lookup failed: {a.query}"); return
663
+ print(f"# Journal metrics, fetched live from OpenAlex - query: {a.query}")
664
+ print(" Note: if2yr = OpenAlex 2yr_mean_citedness, an impact-factor-like measure.\n"
665
+ " It is close to, but not, the official Clarivate IF - use JCR for that, and the\n"
666
+ " journal's own aims & scope page for scope.\n")
667
+ for p in res:
668
+ ifv = f"{p['if2yr']:.1f}" if p.get('if2yr') is not None else "?"
669
+ print(f" {p['name']} [{p.get('type','')}] {p.get('publisher') or ''}")
670
+ print(f" IF-proxy(2yr)={ifv} h-index={p.get('h_index','?')} works={p.get('works','?')} ISSN={p.get('issn','?')} {p.get('homepage') or ''}")
671
+ return
672
+
673
+ if a.cmd == "latest":
674
+ res = latest(a.query, a.n, a.since)
675
+ if a.json:
676
+ print(json.dumps(res, ensure_ascii=False, indent=2)); return
677
+ if not res:
678
+ if _net_failed():
679
+ print(f"Could not query the sources.\n {_net_hint()}"); return
680
+ print(f"No matches since {a.since or CUR_YEAR-1}. Try narrower keywords, "
681
+ "or relax --since."); return
682
+ print(f"# Recent related work (>={a.since or CUR_YEAR-1}, newest first) - query: {a.query}\n")
683
+ for i, p in enumerate(res, 1):
684
+ print(fmt_paper(p, i)); print()
685
+ return
686
+
687
+ if a.cmd == "verify":
688
+ exact = resolve_identifier(a.query)
689
+ if exact:
690
+ if a.json:
691
+ print(json.dumps(exact, ensure_ascii=False, indent=2)); return
692
+ print("MATCH (exact identifier)")
693
+ print(fmt_paper(exact))
694
+ for c in SOURCE_CONFLICTS:
695
+ print(f" ! sources disagree - {c}\n"
696
+ f" arXiv is authoritative for an arXiv id; the record above is theirs.")
697
+ return
698
+ if re.search(DOI_RE, a.query) or re.search(r"arxiv", a.query, re.I):
699
+ # It looked like an identifier and did not resolve - say that,
700
+ # rather than falling back to a title search that cannot succeed.
701
+ if _net_failed():
702
+ print(f"INCONCLUSIVE - could not query the sources: {a.query}\n {_net_hint()}")
703
+ else:
704
+ print(f"NOT FOUND - no record with this identifier: {a.query}\n"
705
+ f" Check the DOI or arXiv id; if it is correct, the work may be too "
706
+ f"new to be indexed.")
707
+ return
708
+ cands = search_openalex(a.query, 3) or []
709
+ if not cands or _match_ratio(a.query, cands[0]["title"]) < 0.6:
710
+ cands = cands + (search_s2(a.query, 3) or [])
711
+ if a.json:
712
+ print(json.dumps(cands, ensure_ascii=False, indent=2)); return
713
+ if not cands:
714
+ # Never call something fake when we simply could not reach the APIs.
715
+ if _net_failed():
716
+ print(f"INCONCLUSIVE - could not query the sources, so nothing can be said "
717
+ f"about: {a.query}\n {_net_hint()}"); return
718
+ print(f"NOT FOUND in any of the four sources -> this citation is very likely "
719
+ f"hallucinated: {a.query}"); return
720
+ best = max(cands, key=lambda p: _match_ratio(a.query, p.get("title", "")))
721
+ r = _match_ratio(a.query, best.get("title", ""))
722
+ verdict = ("MATCH (high confidence)" if r >= 0.75 else
723
+ "PARTIAL MATCH - confirm by hand that this is the same paper" if r >= 0.45 else
724
+ "NO MATCH -> very likely hallucinated")
725
+ print(f"{verdict} [query term coverage = {r:.0%}]")
726
+ print(fmt_paper(best))
727
+ if r < 0.75:
728
+ others = [p for p in cands if p is not best][:3]
729
+ if others:
730
+ print("\nOther candidates:"); [print(fmt_paper(p)) for p in others]
731
+ return
732
+
733
+ if a.cmd == "bibtex":
734
+ b = bibtex(a.query)
735
+ if isinstance(b, tuple) and b and b[0] == "INCONCLUSIVE":
736
+ _, cand, r = b
737
+ print(f"INCONCLUSIVE - a primary source could not be reached, so no entry is emitted "
738
+ f"for: {a.query}")
739
+ print(f" {_net_hint()}")
740
+ if cand:
741
+ print(f" (the partial search's best candidate was {r:.0%} coverage - not enough "
742
+ f"to stand on while sources are down)")
743
+ elif isinstance(b, tuple) and b and b[0] == "WEAK":
744
+ _, cand, r = b
745
+ if cand:
746
+ print(f"No confident match (best term coverage only {r:.0%}). Refusing to emit a "
747
+ f"possibly wrong entry. Closest candidate:")
748
+ print(fmt_paper(cand))
749
+ print('-> If that is the paper, re-run with its DOI: scholarcheck bibtex "<DOI>".\n'
750
+ ' Otherwise give a more exact title.')
751
+ else:
752
+ if _net_failed():
753
+ print(f"Could not query the sources.\n {_net_hint()}")
754
+ else:
755
+ print(f"No candidates resolved - check the spelling: {a.query}")
756
+ elif b:
757
+ print(b)
758
+ else:
759
+ if _net_failed():
760
+ print(f"Could not query the sources.\n {_net_hint()}")
761
+ else:
762
+ print(f"Could not resolve an entry - check the spelling or DOI: {a.query}")
763
+ return
764
+
765
+ if a.cmd == "fetch":
766
+ url, label, stem = resolve_pdf(a.query)
767
+ if not url:
768
+ print(f"❌ {label}"); return
769
+ path = a.out or os.path.join(PDFDIR, (stem or "paper") + ".pdf")
770
+ if download_pdf(url, path):
771
+ print(f"Downloaded [{label}] -> {path}\n"
772
+ f" Next: read it and check the claim against the full text.")
773
+ else:
774
+ print(f"Download failed (flaky network, non-PDF response, or paywalled): {url}\n"
775
+ f" Retry by hand: curl -sL '{url}' -o paper.pdf")
776
+ return
777
+
778
+ if a.cmd == "citedby":
779
+ oid = resolve_openalex_id(a.query)
780
+ if not oid:
781
+ print(f"Could not resolve the target paper (try a title or DOI): {a.query}"); return
782
+ res = citedby_openalex(oid, a.n)
783
+ if a.json:
784
+ print(json.dumps(res or [], ensure_ascii=False, indent=2)); return
785
+ if not res:
786
+ print(f"(OpenAlex {oid}) No citing works recorded yet, or the lookup failed."); return
787
+ print(f"# {len(res)} papers citing {oid} (most cited first) - check whether your\n# extension is already covered:\n")
788
+ for i, p in enumerate(res, 1):
789
+ print(fmt_paper(p, i)); print()
790
+ return
791
+
792
+ def main():
793
+ """CLI entry point. Wraps the real one so the usage nudge cannot change
794
+ the exit status or swallow an exception."""
795
+ try:
796
+ return _main()
797
+ finally:
798
+ from ._nudge import record_run
799
+ record_run()
800
+
801
+
802
+ if __name__ == "__main__":
803
+ main()
@@ -0,0 +1,183 @@
1
+ Metadata-Version: 2.4
2
+ Name: scholarcheck
3
+ Version: 0.1.0
4
+ Summary: Verify citations against real metadata - stop hallucinated references. Zero dependencies.
5
+ Author: Guo Cheng
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/GuoCheng24/scholarcheck
8
+ Project-URL: Issues, https://github.com/GuoCheng24/scholarcheck/issues
9
+ Keywords: citations,bibtex,openalex,crossref,arxiv,literature-review,hallucination,research-tools,doi
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Environment :: Console
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Scientific/Engineering
16
+ Requires-Python: >=3.9
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ Dynamic: license-file
20
+
21
+ # scholarcheck
22
+
23
+ [![test](https://github.com/GuoCheng24/scholarcheck/actions/workflows/test.yml/badge.svg)](https://github.com/GuoCheng24/scholarcheck/actions/workflows/test.yml) [![python](https://img.shields.io/badge/python-3.9%2B-blue)](https://www.python.org/) [![license](https://img.shields.io/badge/license-MIT-green)](LICENSE)
24
+
25
+ **Stop hallucinated citations.** Verify any reference against real metadata — from the command line, with zero dependencies.
26
+
27
+ <p align="center">
28
+ <img src="docs/three-states.png" width="100%">
29
+ </p>
30
+
31
+ <sub>The figure above is generated by <a href="docs/three-states_figure.py">docs/three-states_figure.py</a> — <code>pip install git+https://github.com/GuoCheng24/sciglyph</code> and run it to reproduce <code>docs/three-states.png</code> byte for byte.</sub>
32
+
33
+
34
+ Language models invent plausible-looking papers: right-sounding title, plausible authors, a DOI that resolves to nothing. `scholarcheck` answers one question honestly — **does this paper actually exist?** — by querying OpenAlex, Semantic Scholar, Crossref and arXiv directly.
35
+
36
+ ```console
37
+ $ scholarcheck verify "Deep Residual Learning for Image Recognition"
38
+ MATCH (high confidence) [query term coverage = 100%]
39
+ Deep Residual Learning for Image Recognition (2016, conference-paper; cited=226875) doi:10.1109/cvpr.2016.90
40
+ Kaiming He, Xiangyu Zhang, Shaoqing Ren et al.
41
+
42
+ $ scholarcheck verify "Quantum Topological Radiomics for Zebra Diagnosis in Martian Cohorts"
43
+ NOT FOUND in any of the four sources -> this citation is very likely hallucinated
44
+ ```
45
+
46
+ ## Why not just ask an AI assistant?
47
+
48
+ Because an assistant answers from memory, and memory is exactly what fails here. Three design choices make this different:
49
+
50
+ **1. It says "I could not check" instead of "it is fake."**
51
+ A verifier that reports a network outage as *hallucinated* is worse than no verifier. `scholarcheck` tracks every failed request and distinguishes the two:
52
+
53
+ ```console
54
+ $ scholarcheck verify "Attention Is All You Need" # with the network down
55
+ INCONCLUSIVE - could not query the sources, so nothing can be said about: Attention Is All You Need
56
+ Could not reach: api.openalex.org: curl: (7) Connection refused
57
+ (no proxy set; if your network needs one, set SCHOLARCHECK_PROXY)
58
+ ```
59
+
60
+ It also knows which sources matter: Semantic Scholar rate-limits aggressively without an API key, so its failure never turns a real answer into "inconclusive" — only the primary sources do.
61
+
62
+ **2. It refuses to guess.**
63
+ Ask for BibTeX from a slightly-wrong title and most tools hand back the nearest hit. Silently citing the *wrong* paper is worse than citing none, so a weak match returns the candidate and stops:
64
+
65
+ ```console
66
+ $ scholarcheck bibtex "Deep Residual Learning for Image Recognition in Medicine"
67
+ No confident match (best term coverage only 62%). Refusing to emit a possibly wrong entry.
68
+ Closest candidate:
69
+ Deep Residual Learning for Image Recognition (2016, CVPR) doi:10.1109/CVPR.2016.90
70
+ -> If that is the paper, re-run with its DOI: scholarcheck bibtex "<DOI>".
71
+ ```
72
+
73
+ The same refusal applies when the sources themselves are unavailable, which is
74
+ when a wrong entry is most likely — the "best" match would then be whichever
75
+ paper happened to be reachable:
76
+
77
+ ```console
78
+ $ scholarcheck bibtex "Deep Residual Learning for Image Recognition in Medicine"
79
+ INCONCLUSIVE - a primary source could not be reached, so no entry is emitted for: ...
80
+ Could not reach: api.openalex.org: HTTP 429
81
+ (the partial search's best candidate was 50% coverage - not enough to stand on
82
+ while sources are down)
83
+ ```
84
+
85
+ **3. An identifier is resolved, not searched.**
86
+ `verify "arXiv:1906.08253"` looks the identifier up directly. Feeding it to a
87
+ title matcher would return whatever paper happens to share those digits and
88
+ then score it as a mismatch — which reads as *"this citation is fake"* when the
89
+ truth is that the query was never looked up properly.
90
+
91
+ **4. Recency is a separate command, on purpose.**
92
+ Relevance ranking systematically favours highly-cited older work, which is exactly wrong when you are checking whether someone *just* published your idea. `latest` filters by recency as well as relevance.
93
+
94
+ ## Install
95
+
96
+ ```bash
97
+ pip install git+https://github.com/GuoCheng24/scholarcheck
98
+ ```
99
+
100
+ Or clone and `pip install -e .` if you would rather read the source first — it is
101
+ one file.
102
+
103
+ **No dependencies.** Standard library plus `curl`. Nothing to break, nothing to
104
+ audit, and nothing that needs an API key: every source it queries is open.
105
+
106
+ <sub>Not on PyPI yet, so the git URL above is the install line that works today.
107
+ When it lands, `pip install scholarcheck` will too.</sub>
108
+
109
+ ## Commands
110
+
111
+ | | |
112
+ |---|---|
113
+ | `verify "<title/DOI/arXiv id>"` | Is this citation real? An identifier resolves exactly; a title is matched by term coverage |
114
+ | `bibtex "<DOI/title>"` | A BibTeX entry — refuses to guess on a weak match |
115
+ | `search "<keywords>"` | Multi-source search, re-ranked by term overlap |
116
+ | `latest "<keywords>"` | Recent work only — relevance **and** recency |
117
+ | `priorart "<claim>"` | Nearest N real papers for a claim, plus a checklist for judging whether it is already taken |
118
+ | `citedby "<DOI/title>"` | What cited this paper — has someone already extended it? |
119
+ | `journal "<name>"` | Live journal metrics, instead of quoting an impact factor from memory |
120
+ | `injournal "<name>"` | Recent papers from one journal, to study its actual conventions |
121
+ | `fetch "<DOI/arXiv id>"` | Download the open-access PDF so a claim can be checked in full text |
122
+
123
+ Add `--json` to any command for structured output, `-n` for the number of results, `--since YYYY` to bound the year.
124
+
125
+ ## Use as a library
126
+
127
+ ```python
128
+ from scholarcheck import verify_citation, get_bibtex, NET_ERRORS
129
+
130
+ paper, confidence = verify_citation("Attention Is All You Need")
131
+ if paper is None and NET_ERRORS:
132
+ ... # could not check — not evidence of anything
133
+ elif confidence >= 0.75:
134
+ print(get_bibtex(paper["doi"]))
135
+ ```
136
+
137
+ ## Configuration
138
+
139
+ All optional:
140
+
141
+ | variable | effect |
142
+ |---|---|
143
+ | `SCHOLARCHECK_MAILTO` | your email — joins OpenAlex's polite pool, giving better rate limits |
144
+ | `SCHOLARCHECK_S2KEY` | Semantic Scholar API key (free) — avoids the frequent 429s |
145
+ | `SCHOLARCHECK_PROXY` | e.g. `socks5h://127.0.0.1:1080`; default is a direct connection |
146
+
147
+ Proxy behaviour is decided **solely** by `SCHOLARCHECK_PROXY`. Inherited `http_proxy` / `all_proxy` variables are stripped before each request, so the tool behaves the same on every machine.
148
+
149
+ ## What it can and cannot tell you
150
+
151
+ **A match confirms the paper exists — not that the metadata you have is right.**
152
+ Bibliographic databases often hold several records for one work: a preprint, a
153
+ conference version, a publisher deposit. `verify` returns whichever record
154
+ matched best, so the year and venue you see may belong to a different record
155
+ than the one you meant to cite. Check them; the DOI is the reliable part.
156
+
157
+ **"NOT FOUND" is strong evidence, not proof.** Very new work, non-English
158
+ venues and some book chapters are indexed poorly. When it matters, run
159
+ `search` with looser keywords before concluding a reference is invented.
160
+
161
+ ## Notes from real use
162
+
163
+ - **Feed focused keywords, not whole sentences.** A long claim drags in off-topic papers; two or three precise terms work far better.
164
+ - **`search` favours highly-cited older work.** That is what relevance ranking does. Use `latest` when the question is "has this been done recently?"
165
+ - **A title-only judgement is not a prior-art check.** For the closest candidates, `fetch` the PDF and read it.
166
+
167
+ ## License
168
+
169
+ MIT © Guo Cheng
170
+
171
+ ## 关于那行 star 提示
172
+
173
+ 跑命令时,`scholarcheck` 会在**第 5 次和第 25 次**往 stderr 写一行,提一句这个仓库在哪。**一辈子只有这两次**,此外再不出声。
174
+
175
+ 它不会出现在:管道或重定向里(stderr 不是终端就直接返回,连计数文件都不建)、CI 环境里(`CI` / `GITHUB_ACTIONS`)。它写的是 stderr 而非 stdout,所以不会污染你的数据输出;它包在 `try/finally` 里且吞掉自身所有异常,**不会改变退出码,也不会影响结果**。
176
+
177
+ 永久关掉:
178
+
179
+ ```bash
180
+ export SCHOLARCHECK_NO_NUDGE=1
181
+ ```
182
+
183
+ 计数存在 `$XDG_STATE_HOME/scholarcheck/usage.json`(默认 `~/.local/state/scholarcheck/usage.json`),删掉即重置。
@@ -0,0 +1,9 @@
1
+ scholarcheck/__init__.py,sha256=UCsINEupUkR93_aSX9H-F0T2cb5_47VVemK7owlisgA,1545
2
+ scholarcheck/_nudge.py,sha256=FDpOdxkAObhX2_otyRIDstAxxWup_0tRV7MnqV_m2jU,1897
3
+ scholarcheck/cli.py,sha256=r3gcTLYxmNGA8IBs2Ckkvc5jBxDszmYHo0yz0E9v0ns,38033
4
+ scholarcheck-0.1.0.dist-info/licenses/LICENSE,sha256=7Ujw8dU7FULnBCYz_jKV-VweojL93VUnpQnbHwo6IGk,1066
5
+ scholarcheck-0.1.0.dist-info/METADATA,sha256=aPx6TH4YiZY5ULO6p3r248Ya3mJrnKgLbAdE7ICSJjE,9140
6
+ scholarcheck-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
7
+ scholarcheck-0.1.0.dist-info/entry_points.txt,sha256=niNXVcH0hBfYjSSJcmpSPYr1cWmSsPmz4eUM2wM0FNg,55
8
+ scholarcheck-0.1.0.dist-info/top_level.txt,sha256=SCEMDuN-Vm-QlG85dKk3QZ8Q_JEg5JTsZzd0q5bdJv0,13
9
+ scholarcheck-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ scholarcheck = scholarcheck.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Guo Cheng
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ scholarcheck