scholarcheck 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scholarcheck/__init__.py +50 -0
- scholarcheck/_nudge.py +58 -0
- scholarcheck/cli.py +803 -0
- scholarcheck-0.1.0.dist-info/METADATA +183 -0
- scholarcheck-0.1.0.dist-info/RECORD +9 -0
- scholarcheck-0.1.0.dist-info/WHEEL +5 -0
- scholarcheck-0.1.0.dist-info/entry_points.txt +2 -0
- scholarcheck-0.1.0.dist-info/licenses/LICENSE +21 -0
- scholarcheck-0.1.0.dist-info/top_level.txt +1 -0
scholarcheck/__init__.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""scholarcheck - verifiable literature grounding from the command line.
|
|
2
|
+
|
|
3
|
+
Never cite a paper that does not exist. Every answer is backed by live
|
|
4
|
+
metadata from OpenAlex, Semantic Scholar, Crossref and arXiv.
|
|
5
|
+
|
|
6
|
+
Command line::
|
|
7
|
+
|
|
8
|
+
scholarcheck verify "Attention Is All You Need"
|
|
9
|
+
scholarcheck bibtex "10.1038/s41586-025-10014-0"
|
|
10
|
+
scholarcheck priorart "conformal risk control" -n 6
|
|
11
|
+
|
|
12
|
+
As a library::
|
|
13
|
+
|
|
14
|
+
from scholarcheck import verify_citation, get_bibtex, search
|
|
15
|
+
paper, confidence = verify_citation("Attention Is All You Need")
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from .cli import (
|
|
19
|
+
multi_search as search,
|
|
20
|
+
latest,
|
|
21
|
+
bibtex as get_bibtex,
|
|
22
|
+
best_match,
|
|
23
|
+
citedby_openalex as cited_by,
|
|
24
|
+
journal_lookup as journal,
|
|
25
|
+
injournal,
|
|
26
|
+
resolve_pdf,
|
|
27
|
+
download_pdf,
|
|
28
|
+
NET_ERRORS,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
__version__ = "0.1.0"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def verify_citation(query, n=5):
|
|
35
|
+
"""Check whether a citation refers to a real paper.
|
|
36
|
+
|
|
37
|
+
Returns ``(paper, confidence)``, where confidence is the fraction of the
|
|
38
|
+
query's content words covered by the matched title: >=0.75 is a confident
|
|
39
|
+
match, <0.45 means the citation is very likely hallucinated. ``paper`` is
|
|
40
|
+
None when nothing matched.
|
|
41
|
+
|
|
42
|
+
An empty result together with a non-empty :data:`NET_ERRORS` means the
|
|
43
|
+
sources could not be reached - that is *not* evidence the paper is fake.
|
|
44
|
+
"""
|
|
45
|
+
return best_match(query, n)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
__all__ = ["search", "latest", "get_bibtex", "best_match", "cited_by",
|
|
49
|
+
"journal", "injournal", "resolve_pdf", "download_pdf",
|
|
50
|
+
"verify_citation", "NET_ERRORS", "__version__"]
|
scholarcheck/_nudge.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Mention the repo once or twice, to people who are actually using this.
|
|
2
|
+
|
|
3
|
+
Deliberately quiet: never on the first run, never when stderr is not a
|
|
4
|
+
terminal (so piped and redirected output stays clean), never in CI, and
|
|
5
|
+
never more than twice in the lifetime of an install. `SCHOLARCHECK_NO_NUDGE=1`
|
|
6
|
+
turns it off for good.
|
|
7
|
+
"""
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
REPO = "GuoCheng24/scholarcheck"
|
|
14
|
+
_SHOW_AT = (5, 25) # run counts at which we say something
|
|
15
|
+
_ENV_OFF = "SCHOLARCHECK_NO_NUDGE"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _state_path():
|
|
19
|
+
base = os.environ.get("XDG_STATE_HOME") or (Path.home() / ".local" / "state")
|
|
20
|
+
return Path(base) / "scholarcheck" / "usage.json"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _quiet():
|
|
24
|
+
if os.environ.get(_ENV_OFF):
|
|
25
|
+
return True
|
|
26
|
+
if os.environ.get("CI") or os.environ.get("GITHUB_ACTIONS"):
|
|
27
|
+
return True
|
|
28
|
+
# Not a terminal means someone is piping or redirecting us; stay out of it.
|
|
29
|
+
return not (hasattr(sys.stderr, "isatty") and sys.stderr.isatty())
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def record_run():
|
|
33
|
+
"""Count this run and, at two points, print a single line to stderr.
|
|
34
|
+
|
|
35
|
+
Any failure here is swallowed: a nudge must never break the tool or
|
|
36
|
+
change its exit status.
|
|
37
|
+
"""
|
|
38
|
+
if _quiet():
|
|
39
|
+
return
|
|
40
|
+
try:
|
|
41
|
+
p = _state_path()
|
|
42
|
+
try:
|
|
43
|
+
data = json.loads(p.read_text())
|
|
44
|
+
except Exception:
|
|
45
|
+
data = {}
|
|
46
|
+
n = int(data.get("runs", 0)) + 1
|
|
47
|
+
data["runs"] = n
|
|
48
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
49
|
+
p.write_text(json.dumps(data))
|
|
50
|
+
if n in _SHOW_AT:
|
|
51
|
+
print(
|
|
52
|
+
"\n── scholarcheck has been useful " + str(n) + " times. If it saved you time,\n"
|
|
53
|
+
" a star helps other people find it: https://github.com/" + REPO + "\n"
|
|
54
|
+
" (silence this with SCHOLARCHECK_NO_NUDGE=1)",
|
|
55
|
+
file=sys.stderr,
|
|
56
|
+
)
|
|
57
|
+
except Exception:
|
|
58
|
+
pass
|
scholarcheck/cli.py
ADDED
|
@@ -0,0 +1,803 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""scholarcheck - verifiable literature grounding from the command line.
|
|
3
|
+
|
|
4
|
+
Built for one job: **never cite a paper that does not exist.** Every result
|
|
5
|
+
comes back with real metadata (DOI, authors, year, venue) pulled live from
|
|
6
|
+
public scholarly databases, so a reference can be checked rather than trusted.
|
|
7
|
+
|
|
8
|
+
Sources
|
|
9
|
+
OpenAlex primary, stable, no API key required
|
|
10
|
+
Semantic Scholar citations and TLDRs; often rate-limits without a key
|
|
11
|
+
(set SCHOLARCHECK_S2KEY, free to obtain)
|
|
12
|
+
Crossref DOI -> BibTeX
|
|
13
|
+
arXiv preprints, including theory papers OpenAlex has not indexed
|
|
14
|
+
|
|
15
|
+
Commands
|
|
16
|
+
verify is this citation real? -> match confidence, or "likely hallucinated"
|
|
17
|
+
bibtex DOI / title -> a BibTeX entry (refuses to guess on a weak match)
|
|
18
|
+
search multi-source search, re-ranked by term overlap
|
|
19
|
+
latest recent work only - relevance *and* recency, so new papers are not
|
|
20
|
+
buried under highly-cited old ones
|
|
21
|
+
priorart pull the nearest N real papers for a specific claim, with a
|
|
22
|
+
checklist for judging whether the claim is already taken
|
|
23
|
+
citedby what cited a given paper (has someone already extended it?)
|
|
24
|
+
journal live journal metrics, instead of quoting an impact factor from memory
|
|
25
|
+
injournal recent papers from one journal, to study its actual conventions
|
|
26
|
+
fetch download the open-access PDF so a claim can be checked in full text
|
|
27
|
+
|
|
28
|
+
Environment
|
|
29
|
+
SCHOLARCHECK_PROXY optional proxy, e.g. socks5h://127.0.0.1:1080 (default: direct)
|
|
30
|
+
SCHOLARCHECK_MAILTO your email; joins OpenAlex's polite pool for better rate limits
|
|
31
|
+
SCHOLARCHECK_S2KEY optional Semantic Scholar API key
|
|
32
|
+
|
|
33
|
+
Notes learned the hard way
|
|
34
|
+
* Feed **focused keywords**, not a whole sentence - long claims drag in
|
|
35
|
+
off-topic papers.
|
|
36
|
+
* `search` ranks by relevance and therefore favours highly-cited older work.
|
|
37
|
+
To see what is happening *now*, use `latest`.
|
|
38
|
+
* A weak title match returns nothing rather than a plausible-looking wrong
|
|
39
|
+
entry. Silently citing the wrong paper is worse than citing none.
|
|
40
|
+
"""
|
|
41
|
+
import os, json, subprocess, argparse, urllib.parse, re, time, datetime
|
|
42
|
+
|
|
43
|
+
CUR_YEAR = datetime.date.today().year # resolved at runtime, never hard-coded
|
|
44
|
+
PROXY = os.environ.get("SCHOLARCHECK_PROXY", "") # direct connection by default
|
|
45
|
+
MAILTO = os.environ.get("SCHOLARCHECK_MAILTO", "") # set to join OpenAlex polite pool
|
|
46
|
+
S2KEY = os.environ.get("SCHOLARCHECK_S2KEY")
|
|
47
|
+
DOI_RE = r"10\.\d{4,9}/[-._;()/:A-Za-z0-9]+"
|
|
48
|
+
ARXIV_RE = r"\d{4}\.\d{4,5}(v\d+)?"
|
|
49
|
+
|
|
50
|
+
#: Network failures seen during this run. This matters: without it, a dead
|
|
51
|
+
#: connection looks exactly like "no such paper exists", and a citation
|
|
52
|
+
#: verifier that reports a network outage as "hallucinated" is worse than
|
|
53
|
+
#: useless. Callers check this before concluding anything is fake.
|
|
54
|
+
NET_ERRORS = []
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _clean_env():
|
|
58
|
+
"""Environment for curl, with every inherited *_proxy variable stripped.
|
|
59
|
+
|
|
60
|
+
Proxy behaviour must be decided solely by SCHOLARCHECK_PROXY. Inheriting
|
|
61
|
+
the caller's http_proxy/all_proxy makes the tool behave differently on two
|
|
62
|
+
machines for no visible reason, and mixing an inherited HTTP proxy with an
|
|
63
|
+
explicit SOCKS one fails in ways that are painful to debug.
|
|
64
|
+
"""
|
|
65
|
+
return {k: v for k, v in os.environ.items() if "proxy" not in k.lower()}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _curl(url, accept=None, headers=None, retries=3, timeout=18):
|
|
69
|
+
"""Fetch via curl with backoff on 429/503. Returns the body, or None.
|
|
70
|
+
|
|
71
|
+
On failure the reason is recorded in NET_ERRORS so the caller can tell a
|
|
72
|
+
genuine "not found" apart from "could not reach the API".
|
|
73
|
+
"""
|
|
74
|
+
cmd = ["curl", "-sS", "-L", "--max-time", str(timeout), "-w", "\n__HTTP__%{http_code}"]
|
|
75
|
+
if PROXY:
|
|
76
|
+
cmd += ["-x", PROXY]
|
|
77
|
+
if accept:
|
|
78
|
+
cmd += ["-H", f"Accept: {accept}"]
|
|
79
|
+
for k, v in (headers or {}).items():
|
|
80
|
+
cmd += ["-H", f"{k}: {v}"]
|
|
81
|
+
ua = "scholarcheck/0.1 (+https://github.com/GuoCheng24/scholarcheck)"
|
|
82
|
+
if MAILTO:
|
|
83
|
+
ua += " mailto:%s" % MAILTO
|
|
84
|
+
cmd += ["-H", "User-Agent: " + ua, url]
|
|
85
|
+
host = urllib.parse.urlsplit(url).netloc
|
|
86
|
+
last = "unknown error"
|
|
87
|
+
for k in range(retries):
|
|
88
|
+
try:
|
|
89
|
+
r = subprocess.run(cmd, capture_output=True, text=True,
|
|
90
|
+
env=_clean_env(), timeout=timeout + 6)
|
|
91
|
+
out, code = r.stdout, ""
|
|
92
|
+
if "__HTTP__" in out:
|
|
93
|
+
out, _, code = out.rpartition("__HTTP__"); code = code.strip()
|
|
94
|
+
if r.returncode == 0 and out.strip() and code in ("200", "201", ""):
|
|
95
|
+
return out
|
|
96
|
+
if r.returncode != 0:
|
|
97
|
+
last = (r.stderr or "").strip().splitlines()[-1] if r.stderr else f"curl exit {r.returncode}"
|
|
98
|
+
elif code:
|
|
99
|
+
last = f"HTTP {code}"
|
|
100
|
+
if code in ("429", "403", "503", "500", "502"): # rate-limited or temporarily unavailable -> back off
|
|
101
|
+
time.sleep(2.0 * (k + 1)); continue
|
|
102
|
+
except Exception as e:
|
|
103
|
+
last = f"{type(e).__name__}: {e}"
|
|
104
|
+
time.sleep(1.0 * (k + 1))
|
|
105
|
+
NET_ERRORS.append(f"{host}: {last}")
|
|
106
|
+
return None
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
#: Sources whose failure genuinely means "we could not look". Semantic Scholar
|
|
110
|
+
#: rate-limits hard without an API key and arXiv only supplements coverage, so
|
|
111
|
+
#: neither should turn a real answer into "inconclusive" - otherwise the tool
|
|
112
|
+
#: would refuse to flag anything whenever S2 returns 429, which is often.
|
|
113
|
+
_PRIMARY = ("openalex", "crossref")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _net_failed(primary_only=True):
|
|
117
|
+
"""True if a source we actually depend on could not be reached."""
|
|
118
|
+
if primary_only:
|
|
119
|
+
return any(any(h in e for h in _PRIMARY) for e in NET_ERRORS)
|
|
120
|
+
return bool(NET_ERRORS)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _net_hint():
|
|
124
|
+
"""A one-line, actionable explanation of what went wrong on the network."""
|
|
125
|
+
crit = [e for e in NET_ERRORS if any(h in e for h in _PRIMARY)]
|
|
126
|
+
uniq = list(dict.fromkeys(crit or NET_ERRORS))[:3]
|
|
127
|
+
hint = "Could not reach: " + "; ".join(uniq)
|
|
128
|
+
if PROXY:
|
|
129
|
+
hint += f"\n (using proxy {PROXY} from SCHOLARCHECK_PROXY - check it is reachable)"
|
|
130
|
+
else:
|
|
131
|
+
hint += "\n (no proxy set; if your network needs one, set SCHOLARCHECK_PROXY)"
|
|
132
|
+
return hint
|
|
133
|
+
|
|
134
|
+
def _get_json(url, headers=None):
|
|
135
|
+
t = _curl(url, accept="application/json", headers=headers)
|
|
136
|
+
if not t:
|
|
137
|
+
return None
|
|
138
|
+
try:
|
|
139
|
+
return json.loads(t)
|
|
140
|
+
except Exception:
|
|
141
|
+
return None
|
|
142
|
+
|
|
143
|
+
# ---------- text utilities ----------
|
|
144
|
+
_STOP = set("the a an of for and or to in on with via using from is are be that this we our by as at "
|
|
145
|
+
"how when what which under over into onto not no".split())
|
|
146
|
+
def _tokens(s, minlen=4):
|
|
147
|
+
return [w for w in re.findall(r"[a-zA-Z][a-zA-Z\-]+", (s or "").lower()) if len(w) >= minlen and w not in _STOP]
|
|
148
|
+
def _match_ratio(query, title):
|
|
149
|
+
"""Fraction of the query's content words covered by the title (used by `verify`)."""
|
|
150
|
+
tq = set(_tokens(query, 3)); tt = set(_tokens(title, 3))
|
|
151
|
+
if not tq or not tt:
|
|
152
|
+
return 0.0
|
|
153
|
+
return len(tq & tt) / len(tq)
|
|
154
|
+
def _relevance(query, p):
|
|
155
|
+
"""Term hits weighted title x3 + abstract x1; used to re-rank away off-topic results."""
|
|
156
|
+
toks = set(_tokens(query))
|
|
157
|
+
title = (p.get("title") or "").lower(); ab = (p.get("abstract") or "").lower()
|
|
158
|
+
return sum((3 if w in title else 0) + (1 if w in ab else 0) for w in toks)
|
|
159
|
+
|
|
160
|
+
# ---------- OpenAlex ----------
|
|
161
|
+
def _oa_abstract(inv):
|
|
162
|
+
if not inv:
|
|
163
|
+
return ""
|
|
164
|
+
pos = {}
|
|
165
|
+
for w, idxs in inv.items():
|
|
166
|
+
for i in idxs:
|
|
167
|
+
pos[i] = w
|
|
168
|
+
s = " ".join(pos[i] for i in sorted(pos))
|
|
169
|
+
return (s[:300] + "…") if len(s) > 300 else s
|
|
170
|
+
|
|
171
|
+
def _oa_work(w):
|
|
172
|
+
src = (w.get("primary_location") or {}).get("source") or {}
|
|
173
|
+
ids = w.get("ids") or {}
|
|
174
|
+
return {
|
|
175
|
+
"title": w.get("title") or "(no title)",
|
|
176
|
+
"year": w.get("publication_year"),
|
|
177
|
+
"venue": src.get("display_name") or (w.get("type") or ""),
|
|
178
|
+
"doi": (w.get("doi") or "").replace("https://doi.org/", "") or None,
|
|
179
|
+
"arxiv": None,
|
|
180
|
+
"id": (w.get("id") or "").replace("https://openalex.org/", ""),
|
|
181
|
+
"cited_by": w.get("cited_by_count", 0),
|
|
182
|
+
"authors": [a.get("author", {}).get("display_name") for a in (w.get("authorships") or [])[:6]],
|
|
183
|
+
"abstract": _oa_abstract(w.get("abstract_inverted_index")),
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
def search_openalex(query, n, since=None):
|
|
187
|
+
q = urllib.parse.quote(query)
|
|
188
|
+
url = f"https://api.openalex.org/works?search={q}&per-page={n}&sort=relevance_score:desc&mailto={MAILTO}"
|
|
189
|
+
if since:
|
|
190
|
+
url += f"&filter=from_publication_date:{since}-01-01"
|
|
191
|
+
d = _get_json(url)
|
|
192
|
+
if not d or "results" not in d:
|
|
193
|
+
return None
|
|
194
|
+
return [_oa_work(w) for w in d["results"]]
|
|
195
|
+
|
|
196
|
+
def citedby_openalex(oa_id, n):
|
|
197
|
+
url = f"https://api.openalex.org/works?filter=cites:{oa_id}&per-page={n}&sort=cited_by_count:desc&mailto={MAILTO}"
|
|
198
|
+
d = _get_json(url)
|
|
199
|
+
if not d or "results" not in d:
|
|
200
|
+
return None
|
|
201
|
+
return [_oa_work(w) for w in d["results"]]
|
|
202
|
+
|
|
203
|
+
def journal_lookup(name, n=5):
|
|
204
|
+
"""Live journal metrics. OpenAlex 2yr_mean_citedness is an IF-like measure:
|
|
205
|
+
open, current, and not behind the JCR paywall - but not the official IF."""
|
|
206
|
+
q = urllib.parse.quote(name)
|
|
207
|
+
d = _get_json(f"https://api.openalex.org/sources?search={q}&per-page={n}&mailto={MAILTO}")
|
|
208
|
+
if not d or "results" not in d:
|
|
209
|
+
return None
|
|
210
|
+
out = []
|
|
211
|
+
for s in d["results"]:
|
|
212
|
+
ss = s.get("summary_stats") or {}
|
|
213
|
+
out.append({
|
|
214
|
+
"name": s.get("display_name"), "type": s.get("type"),
|
|
215
|
+
"id": (s.get("id") or "").replace("https://openalex.org/", ""),
|
|
216
|
+
"if2yr": ss.get("2yr_mean_citedness"), "h_index": ss.get("h_index"),
|
|
217
|
+
"works": s.get("works_count"), "issn": s.get("issn_l"),
|
|
218
|
+
"publisher": s.get("host_organization_name"), "homepage": s.get("homepage_url"),
|
|
219
|
+
})
|
|
220
|
+
return out
|
|
221
|
+
|
|
222
|
+
def injournal(name, n, topic=None, since=None):
|
|
223
|
+
"""Recent papers from one journal, to study its actual conventions. Returns (name, papers)."""
|
|
224
|
+
js = journal_lookup(name, 1)
|
|
225
|
+
if not js or not js[0].get("id"):
|
|
226
|
+
return None, None
|
|
227
|
+
sid = js[0]["id"]; jname = js[0]["name"]
|
|
228
|
+
yr = int(since) if since else CUR_YEAR - 2
|
|
229
|
+
url = (f"https://api.openalex.org/works?filter=primary_location.source.id:{sid},"
|
|
230
|
+
f"from_publication_date:{yr}-01-01&per-page={n * 3}&sort=publication_date:desc&mailto={MAILTO}")
|
|
231
|
+
if topic:
|
|
232
|
+
url += f"&search={urllib.parse.quote(topic)}"
|
|
233
|
+
d = _get_json(url)
|
|
234
|
+
if not d or "results" not in d:
|
|
235
|
+
return jname, None
|
|
236
|
+
ws = []
|
|
237
|
+
for w in d["results"]:
|
|
238
|
+
p = _oa_work(w)
|
|
239
|
+
p["is_oa"] = bool((w.get("open_access") or {}).get("is_oa"))
|
|
240
|
+
ws.append(p)
|
|
241
|
+
if topic:
|
|
242
|
+
ws.sort(key=lambda p: (_relevance(topic, p), p.get("year") or 0), reverse=True)
|
|
243
|
+
return jname, ws[:n]
|
|
244
|
+
|
|
245
|
+
def resolve_openalex_id(s):
|
|
246
|
+
"""Title / DOI / OpenAlex id -> an OpenAlex work id (used by `citedby`)."""
|
|
247
|
+
s = s.strip()
|
|
248
|
+
if re.fullmatch(r"W\d+", s):
|
|
249
|
+
return s
|
|
250
|
+
m = re.search(DOI_RE, s)
|
|
251
|
+
if m:
|
|
252
|
+
d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(m.group(0))}&per-page=1&mailto={MAILTO}")
|
|
253
|
+
if d and d.get("results"):
|
|
254
|
+
return d["results"][0]["id"].replace("https://openalex.org/", "")
|
|
255
|
+
best, r = best_match(s) # titles go through best_match; too weak -> refuse to resolve
|
|
256
|
+
if not best or r < 0.5:
|
|
257
|
+
return None
|
|
258
|
+
if (best.get("id") or "").startswith("W"):
|
|
259
|
+
return best["id"]
|
|
260
|
+
if best.get("doi"): # matched in S2/arXiv -> exchange the DOI for an OpenAlex id
|
|
261
|
+
d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(best['doi'])}&per-page=1&mailto={MAILTO}")
|
|
262
|
+
if d and d.get("results"):
|
|
263
|
+
return d["results"][0]["id"].replace("https://openalex.org/", "")
|
|
264
|
+
return None
|
|
265
|
+
|
|
266
|
+
# ---------- Semantic Scholar ----------
|
|
267
|
+
def search_s2(query, n):
|
|
268
|
+
q = urllib.parse.quote(query)
|
|
269
|
+
fields = "title,year,venue,authors,externalIds,abstract,citationCount,tldr"
|
|
270
|
+
url = f"https://api.semanticscholar.org/graph/v1/paper/search?query={q}&limit={n}&fields={fields}"
|
|
271
|
+
d = _get_json(url, headers=({"x-api-key": S2KEY} if S2KEY else None))
|
|
272
|
+
if not d or "data" not in d:
|
|
273
|
+
return None
|
|
274
|
+
out = []
|
|
275
|
+
for p in d["data"]:
|
|
276
|
+
ext = p.get("externalIds") or {}
|
|
277
|
+
out.append({
|
|
278
|
+
"title": p.get("title"), "year": p.get("year"),
|
|
279
|
+
"venue": p.get("venue") or "", "doi": ext.get("DOI"),
|
|
280
|
+
"arxiv": ext.get("ArXiv"), "id": p.get("paperId"),
|
|
281
|
+
"cited_by": p.get("citationCount", 0),
|
|
282
|
+
"authors": [a.get("name") for a in (p.get("authors") or [])[:6]],
|
|
283
|
+
"abstract": (p.get("tldr") or {}).get("text") or (p.get("abstract") or "")[:300],
|
|
284
|
+
})
|
|
285
|
+
return out
|
|
286
|
+
|
|
287
|
+
# ---------- arXiv ----------
|
|
288
|
+
def search_arxiv(query, n):
|
|
289
|
+
import xml.etree.ElementTree as ET
|
|
290
|
+
q = urllib.parse.quote(query)
|
|
291
|
+
url = f"https://export.arxiv.org/api/query?search_query=all:{q}&start=0&max_results={n}&sortBy=relevance"
|
|
292
|
+
t = _curl(url)
|
|
293
|
+
if not t:
|
|
294
|
+
return None
|
|
295
|
+
try:
|
|
296
|
+
root = ET.fromstring(t)
|
|
297
|
+
except Exception:
|
|
298
|
+
return None
|
|
299
|
+
ns = {"a": "http://www.w3.org/2005/Atom"}
|
|
300
|
+
out = []
|
|
301
|
+
for e in root.findall("a:entry", ns):
|
|
302
|
+
aid = (e.findtext("a:id", default="", namespaces=ns) or "").split("/abs/")[-1]
|
|
303
|
+
yr = (e.findtext("a:published", default="", namespaces=ns) or "")[:4]
|
|
304
|
+
out.append({
|
|
305
|
+
"title": " ".join((e.findtext("a:title", default="", namespaces=ns) or "").split()),
|
|
306
|
+
"year": int(yr) if yr.isdigit() else None,
|
|
307
|
+
"venue": "arXiv", "doi": None, "arxiv": aid, "id": aid, "cited_by": 0,
|
|
308
|
+
"authors": [a.findtext("a:name", default="", namespaces=ns) for a in e.findall("a:author", ns)][:6],
|
|
309
|
+
"abstract": " ".join((e.findtext("a:summary", default="", namespaces=ns) or "").split())[:300],
|
|
310
|
+
})
|
|
311
|
+
return out
|
|
312
|
+
|
|
313
|
+
def arxiv_by_id(aid):
|
|
314
|
+
"""Look an arXiv id up on arXiv itself. Returns one paper dict, or None.
|
|
315
|
+
|
|
316
|
+
This is the authoritative mapping for an arXiv id, and it is used in
|
|
317
|
+
preference to resolving the id through a DOI. Aggregators derive the
|
|
318
|
+
10.48550/arXiv.* DOI second-hand and can attach it to the wrong record -
|
|
319
|
+
observed in the wild, returning an unrelated paper for a valid id, which is
|
|
320
|
+
far more damaging than returning nothing.
|
|
321
|
+
"""
|
|
322
|
+
import xml.etree.ElementTree as ET
|
|
323
|
+
|
|
324
|
+
aid = aid.split("v")[0]
|
|
325
|
+
t = _curl(f"https://export.arxiv.org/api/query?id_list={urllib.parse.quote(aid)}")
|
|
326
|
+
if not t:
|
|
327
|
+
return None
|
|
328
|
+
try:
|
|
329
|
+
root = ET.fromstring(t)
|
|
330
|
+
except Exception:
|
|
331
|
+
return None
|
|
332
|
+
ns = {"a": "http://www.w3.org/2005/Atom"}
|
|
333
|
+
e = root.find("a:entry", ns)
|
|
334
|
+
if e is None:
|
|
335
|
+
return None
|
|
336
|
+
title = " ".join((e.findtext("a:title", default="", namespaces=ns) or "").split())
|
|
337
|
+
if not title or title.lower().startswith("error"):
|
|
338
|
+
return None
|
|
339
|
+
yr = (e.findtext("a:published", default="", namespaces=ns) or "")[:4]
|
|
340
|
+
doi = e.findtext("a:doi", default="", namespaces=ns) or None
|
|
341
|
+
return {
|
|
342
|
+
"title": title,
|
|
343
|
+
"year": int(yr) if yr.isdigit() else None,
|
|
344
|
+
"venue": "arXiv", "doi": doi, "arxiv": aid, "id": aid, "cited_by": 0,
|
|
345
|
+
"authors": [a.findtext("a:name", default="", namespaces=ns)
|
|
346
|
+
for a in e.findall("a:author", ns)][:6],
|
|
347
|
+
"abstract": " ".join((e.findtext("a:summary", default="", namespaces=ns) or "").split())[:300],
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
#: Recorded when two sources return materially different records for the same
|
|
352
|
+
#: identifier. Cross-checking is the whole point of querying more than one.
|
|
353
|
+
SOURCE_CONFLICTS = []
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
# ---------- BibTeX ----------
|
|
357
|
+
def _fallback_bibtex(p):
|
|
358
|
+
"""Build an entry from metadata when there is no Crossref DOI (arXiv -> @misc with eprint)."""
|
|
359
|
+
names = [x for x in (p.get("authors") or []) if x]
|
|
360
|
+
auth = " and ".join(names) if names else "Unknown"
|
|
361
|
+
y = p.get("year") or "n.d."
|
|
362
|
+
first = (re.sub(r"[^A-Za-z]", "", (names[0].split()[-1] if names else "anon")) or "anon").lower()
|
|
363
|
+
kw = (_tokens(p.get("title", "")) or ["ref"])[0]
|
|
364
|
+
key = f"{first}{y}{kw}"
|
|
365
|
+
title = p.get("title", "")
|
|
366
|
+
if p.get("arxiv"):
|
|
367
|
+
return (f"@misc{{{key},\n title={{{title}}},\n author={{{auth}}},\n year={{{y}}},\n"
|
|
368
|
+
f" eprint={{{p['arxiv']}}},\n archivePrefix={{arXiv}},\n note={{arXiv:{p['arxiv']}}}\n}}")
|
|
369
|
+
return (f"@article{{{key},\n title={{{title}}},\n author={{{auth}}},\n year={{{y}}},\n"
|
|
370
|
+
f" journal={{{p.get('venue','')}}}\n}}")
|
|
371
|
+
|
|
372
|
+
def bibtex(s):
|
|
373
|
+
"""Returns a BibTeX string; ('WEAK', candidate, ratio) when the title match is
|
|
374
|
+
too weak to be safe; ('INCONCLUSIVE', candidate, ratio) when a primary source
|
|
375
|
+
could not be reached; or None. It will not hand back a wrong entry."""
|
|
376
|
+
s = s.strip()
|
|
377
|
+
m = re.search(DOI_RE, s)
|
|
378
|
+
doi = m.group(0) if m else None
|
|
379
|
+
meta = None
|
|
380
|
+
if not doi: # title input: best_match guards against returning the wrong paper
|
|
381
|
+
NET_ERRORS.clear()
|
|
382
|
+
meta, r = best_match(s)
|
|
383
|
+
# A degraded search is exactly when the wrong entry is most likely: the
|
|
384
|
+
# "best" match is then drawn from whichever sources happened to answer.
|
|
385
|
+
# `verify` already refuses to speak under those conditions; emitting a
|
|
386
|
+
# citation would be a stronger claim than refusing to make a weaker one.
|
|
387
|
+
if _net_failed():
|
|
388
|
+
return ("INCONCLUSIVE", meta, r)
|
|
389
|
+
# Inclusive boundary: half the query's content words is not a confident
|
|
390
|
+
# match by any reading, and an exact 0.50 used to slip through.
|
|
391
|
+
if not meta or r <= 0.5:
|
|
392
|
+
return ("WEAK", meta, r) # better to return nothing than to silently emit a wrong citation
|
|
393
|
+
doi = meta.get("doi")
|
|
394
|
+
if doi:
|
|
395
|
+
bib = _curl(f"https://doi.org/{doi}", accept="application/x-bibtex")
|
|
396
|
+
if bib and "@" in bib:
|
|
397
|
+
return bib.strip()
|
|
398
|
+
if meta is None and doi: # Crossref failed -> fall back to OpenAlex metadata
|
|
399
|
+
d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(doi)}&per-page=1&mailto={MAILTO}")
|
|
400
|
+
if d and d.get("results"):
|
|
401
|
+
meta = _oa_work(d["results"][0])
|
|
402
|
+
return _fallback_bibtex(meta) if meta else None
|
|
403
|
+
|
|
404
|
+
# ---------- PDF retrieval: from a hit to the actual full text ----------
|
|
405
|
+
PDFDIR = os.environ.get("SCHOLAR_PDFDIR", "/tmp/scholar_pdfs")
|
|
406
|
+
|
|
407
|
+
def _openalex_work_full(doi=None, oa_id=None):
|
|
408
|
+
if doi:
|
|
409
|
+
d = _get_json(f"https://api.openalex.org/works?filter=doi:{urllib.parse.quote(doi)}&per-page=1&mailto={MAILTO}")
|
|
410
|
+
return (d.get("results") or [None])[0] if d else None
|
|
411
|
+
if oa_id:
|
|
412
|
+
return _get_json(f"https://api.openalex.org/works/{oa_id}?mailto={MAILTO}")
|
|
413
|
+
return None
|
|
414
|
+
|
|
415
|
+
def _work_pdf_url(work):
|
|
416
|
+
"""Find a downloadable PDF on an OpenAlex work: arXiv first, then OA pdf_url/oa_url."""
|
|
417
|
+
if not work:
|
|
418
|
+
return None, None
|
|
419
|
+
for loc in [work.get("primary_location")] + (work.get("locations") or []):
|
|
420
|
+
if not loc:
|
|
421
|
+
continue
|
|
422
|
+
landing = loc.get("landing_page_url") or ""
|
|
423
|
+
host = ((loc.get("source") or {}).get("display_name") or "")
|
|
424
|
+
if "arxiv" in (landing + host).lower():
|
|
425
|
+
m = re.search(r"(\d{4}\.\d{4,5})", landing)
|
|
426
|
+
if m:
|
|
427
|
+
return f"https://arxiv.org/pdf/{m.group(1)}", "arXiv:" + m.group(1)
|
|
428
|
+
if loc.get("pdf_url"):
|
|
429
|
+
return loc["pdf_url"], "OA-pdf"
|
|
430
|
+
oa = (work.get("open_access") or {}).get("oa_url")
|
|
431
|
+
return (oa, "OA") if oa else (None, None)
|
|
432
|
+
|
|
433
|
+
def resolve_pdf(s):
|
|
434
|
+
"""arXiv id / DOI / title / URL -> (pdf_url, label, stem) or (None, reason, None)."""
|
|
435
|
+
s = s.strip()
|
|
436
|
+
m = re.fullmatch(r"(?:arxiv:)?(\d{4}\.\d{4,5})(v\d+)?", s, re.I)
|
|
437
|
+
if m:
|
|
438
|
+
aid = m.group(1) + (m.group(2) or "")
|
|
439
|
+
return f"https://arxiv.org/pdf/{aid}", "arXiv:" + aid, aid.replace(".", "_")
|
|
440
|
+
if s.lower().startswith("http"):
|
|
441
|
+
return s, "url", re.sub(r"\W+", "_", s)[-40:]
|
|
442
|
+
doi_m = re.search(DOI_RE, s)
|
|
443
|
+
work, stem = None, "paper"
|
|
444
|
+
if doi_m:
|
|
445
|
+
work = _openalex_work_full(doi=doi_m.group(0)); stem = doi_m.group(0).replace("/", "_")
|
|
446
|
+
else:
|
|
447
|
+
best, r = best_match(s)
|
|
448
|
+
if not best or r < 0.5:
|
|
449
|
+
return None, "title match too weak - use a DOI, an arXiv id, or a more exact title", None
|
|
450
|
+
stem = re.sub(r"\W+", "_", (best.get("title") or "paper"))[:40]
|
|
451
|
+
if best.get("arxiv"):
|
|
452
|
+
aid = best["arxiv"]
|
|
453
|
+
return f"https://arxiv.org/pdf/{aid}", "arXiv:" + aid, aid.replace(".", "_")
|
|
454
|
+
if (best.get("id") or "").startswith("W"):
|
|
455
|
+
work = _openalex_work_full(oa_id=best["id"])
|
|
456
|
+
elif best.get("doi"):
|
|
457
|
+
work = _openalex_work_full(doi=best["doi"])
|
|
458
|
+
url, label = _work_pdf_url(work)
|
|
459
|
+
return (url, label, stem) if url else (None, "no open-access PDF found (likely paywalled - try the publisher or your library)", None)
|
|
460
|
+
|
|
461
|
+
def download_pdf(url, path, retries=3):
|
|
462
|
+
os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
|
|
463
|
+
env = dict(os.environ); env["no_proxy"] = ""; env["NO_PROXY"] = ""
|
|
464
|
+
cands = [url]
|
|
465
|
+
m = re.search(r"PMC(\d+)", url) # direct PMC links are often blocked -> fall back to the europepmc renderer
|
|
466
|
+
if m:
|
|
467
|
+
cands.append(f"https://europepmc.org/articles/PMC{m.group(1)}?pdf=render")
|
|
468
|
+
UA = "Mozilla/5.0 (X11; Linux x86_64) scholar-ground/1.0"
|
|
469
|
+
for u in cands:
|
|
470
|
+
for k in range(retries): # these endpoints are flaky; retry automatically
|
|
471
|
+
try:
|
|
472
|
+
subprocess.run(["curl", "-sL", "--max-time", "60", "-x", PROXY,
|
|
473
|
+
"-H", f"User-Agent: {UA}", "-o", path, u],
|
|
474
|
+
env=env, timeout=70, capture_output=True)
|
|
475
|
+
if os.path.getsize(path) > 1000 and open(path, "rb").read(5) == b"%PDF-":
|
|
476
|
+
return True
|
|
477
|
+
except Exception:
|
|
478
|
+
pass
|
|
479
|
+
time.sleep(1.2 * (k + 1))
|
|
480
|
+
try: # drop non-PDF junk (error pages) instead of leaving a broken file
|
|
481
|
+
if os.path.exists(path) and open(path, "rb").read(5) != b"%PDF-":
|
|
482
|
+
os.remove(path)
|
|
483
|
+
except Exception:
|
|
484
|
+
pass
|
|
485
|
+
return False
|
|
486
|
+
|
|
487
|
+
# ---------- output ----------
|
|
488
|
+
def _locator(p):
|
|
489
|
+
if p.get("doi"):
|
|
490
|
+
return "doi:" + p["doi"]
|
|
491
|
+
if p.get("arxiv"):
|
|
492
|
+
return "arXiv:" + p["arxiv"]
|
|
493
|
+
pid = p.get("id") or ""
|
|
494
|
+
if pid.startswith("W"):
|
|
495
|
+
return "OpenAlex:" + pid
|
|
496
|
+
return "no locator"
|
|
497
|
+
|
|
498
|
+
def fmt_paper(p, i=None):
|
|
499
|
+
tag = f"[{i}] " if i is not None else ""
|
|
500
|
+
names = [x for x in (p.get("authors") or []) if x]
|
|
501
|
+
au = ", ".join(names[:3]) + (" et al." if len(names) > 3 else "")
|
|
502
|
+
head = f"{tag}{p.get('title') or '(no title)'} ({p.get('year','?')}, {p.get('venue') or '?'}; cited={p.get('cited_by',0)}) {_locator(p)}"
|
|
503
|
+
body = (f" {au}\n {p.get('abstract','')}").rstrip()
|
|
504
|
+
return head + ("\n" + body if body.strip() else "")
|
|
505
|
+
|
|
506
|
+
def multi_search(query, n, since=None):
|
|
507
|
+
"""OpenAlex as the stable base, opportunistically enriched with S2 and arXiv,
|
|
508
|
+
de-duplicated, then re-ranked by term overlap."""
|
|
509
|
+
pool, seen = [], set()
|
|
510
|
+
def add(lst):
|
|
511
|
+
for p in (lst or []):
|
|
512
|
+
k = (p.get("title") or "").lower().strip()[:60]
|
|
513
|
+
if k and k not in seen:
|
|
514
|
+
seen.add(k); pool.append(p)
|
|
515
|
+
add(search_openalex(query, n * 3, since))
|
|
516
|
+
add(search_s2(query, n))
|
|
517
|
+
if len(pool) < n:
|
|
518
|
+
add(search_arxiv(query, n))
|
|
519
|
+
if since:
|
|
520
|
+
yr0 = int(since)
|
|
521
|
+
pool = [p for p in pool if (p.get("year") or 0) >= yr0] or pool
|
|
522
|
+
pool.sort(key=lambda p: (_relevance(query, p) + _year_bonus(p), p.get("cited_by", 0)), reverse=True)
|
|
523
|
+
return pool[:n]
|
|
524
|
+
|
|
525
|
+
def _year_bonus(p):
|
|
526
|
+
"""Recency weighting, so this year's papers are not buried under highly-cited old ones."""
|
|
527
|
+
y = p.get("year") or 0
|
|
528
|
+
if y >= CUR_YEAR: return 2.5
|
|
529
|
+
if y >= CUR_YEAR - 1: return 1.3
|
|
530
|
+
if y >= CUR_YEAR - 2: return 0.5
|
|
531
|
+
return 0.0
|
|
532
|
+
|
|
533
|
+
def latest(query, n, since=None):
|
|
534
|
+
"""Recent related work: relevance ranking plus a recency filter, which avoids the
|
|
535
|
+
noise of sorting by date alone. Defaults to roughly the last 18 months."""
|
|
536
|
+
yr = int(since) if since else CUR_YEAR - 1
|
|
537
|
+
res = (search_openalex(query, n * 2, since=str(yr)) or [])
|
|
538
|
+
s2 = search_s2(query, n) or []
|
|
539
|
+
seen = {(p.get("title") or "").lower()[:60] for p in res}
|
|
540
|
+
for p in s2:
|
|
541
|
+
if (p.get("year") or 0) >= yr and (p.get("title") or "").lower()[:60] not in seen:
|
|
542
|
+
res.append(p)
|
|
543
|
+
res.sort(key=lambda p: (p.get("year") or 0, _relevance(query, p)), reverse=True)
|
|
544
|
+
return res[:n]
|
|
545
|
+
|
|
546
|
+
def resolve_identifier(s):
|
|
547
|
+
"""A DOI or arXiv id resolved exactly, or None if `s` is not an identifier.
|
|
548
|
+
|
|
549
|
+
Identifiers must not go through title search. Feeding "arXiv:1906.08253"
|
|
550
|
+
to a title matcher returns whatever paper happens to share those digits and
|
|
551
|
+
then scores it as a mismatch - which reads as "this citation is fake" when
|
|
552
|
+
the truth is that the query was never looked up properly.
|
|
553
|
+
"""
|
|
554
|
+
m = re.search(DOI_RE, s)
|
|
555
|
+
if m:
|
|
556
|
+
w = _openalex_work_full(doi=m.group(0))
|
|
557
|
+
return _oa_work(w) if w else None
|
|
558
|
+
|
|
559
|
+
m = re.search(r"(?:arxiv[:\s/]*)?(" + ARXIV_RE + r")", s, re.I)
|
|
560
|
+
if m and (re.search(r"arxiv", s, re.I) or re.fullmatch(r"[\d.v]+", s.strip())):
|
|
561
|
+
aid = m.group(1)
|
|
562
|
+
primary = arxiv_by_id(aid) # authoritative for an arXiv id
|
|
563
|
+
w = _openalex_work_full(doi="10.48550/arXiv." + aid.split("v")[0])
|
|
564
|
+
secondary = _oa_work(w) if w else None
|
|
565
|
+
if primary and secondary:
|
|
566
|
+
# Disagreement means an aggregator has the id attached to the wrong
|
|
567
|
+
# work. Trust arXiv, and say so rather than silently picking one.
|
|
568
|
+
if _match_ratio(primary["title"], secondary["title"]) < 0.5:
|
|
569
|
+
SOURCE_CONFLICTS.append(
|
|
570
|
+
f"arXiv:{aid} -> arXiv says \"{primary['title'][:60]}\"; "
|
|
571
|
+
f"the aggregator says \"{secondary['title'][:60]}\"")
|
|
572
|
+
else:
|
|
573
|
+
primary["cited_by"] = secondary.get("cited_by", 0) or 0
|
|
574
|
+
primary["venue"] = secondary.get("venue") or primary["venue"]
|
|
575
|
+
return primary or secondary
|
|
576
|
+
return None
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def best_match(query, n=6):
|
|
580
|
+
"""Title -> best matching paper, with re-ranking and a coverage ratio, so bibtex
|
|
581
|
+
and citedby never silently resolve to the wrong paper. Returns (paper|None, ratio)."""
|
|
582
|
+
cands = multi_search(query, n) or []
|
|
583
|
+
best = max(cands, key=lambda p: _match_ratio(query, p.get("title", "")), default=None)
|
|
584
|
+
r = _match_ratio(query, best.get("title", "")) if best else 0.0
|
|
585
|
+
if r < 0.6: # weak match -> also try arXiv directly (theory papers are often arXiv-only)
|
|
586
|
+
for p in (search_arxiv(query, 5) or []):
|
|
587
|
+
rr = _match_ratio(query, p.get("title", ""))
|
|
588
|
+
if rr > r:
|
|
589
|
+
best, r = p, rr
|
|
590
|
+
return best, r
|
|
591
|
+
|
|
592
|
+
OCC_TEMPLATE = (
|
|
593
|
+
"—" * 60 + "\n"
|
|
594
|
+
"[Prior-art check] Ask this of every paper below. All 'no' => the claim is still open:\n"
|
|
595
|
+
" * Does it do *exactly* your specific twist and mechanism? Overlapping on the\n"
|
|
596
|
+
" broader topic alone does not count as taken.\n"
|
|
597
|
+
" * Does it cover the dimension your claim is more specific about - which\n"
|
|
598
|
+
" variable, regime, bound or mechanism?\n"
|
|
599
|
+
" * If one looks close, run `citedby \"<its DOI/title>\"` to see whether a\n"
|
|
600
|
+
" follow-up already covers your extension.\n"
|
|
601
|
+
" ! Never conclude 'already taken' from a title alone. For the closest papers,\n"
|
|
602
|
+
" run `scholarcheck fetch \"<DOI/arXiv-id/title>\"` and check the full text."
|
|
603
|
+
)
|
|
604
|
+
|
|
605
|
+
def _main():
|
|
606
|
+
ap = argparse.ArgumentParser(
|
|
607
|
+
description="Verifiable literature grounding - OpenAlex / Semantic Scholar / Crossref / arXiv",
|
|
608
|
+
epilog='example: scholarcheck priorart "low-degree polynomial detection lower bound" -n 6 --since 2020',
|
|
609
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
610
|
+
ap.add_argument("cmd", choices=["search", "priorart", "occupancy", "verify", "bibtex", "citedby", "fetch", "latest", "journal", "injournal"])
|
|
611
|
+
ap.add_argument("query", help="focused keywords / paper title / DOI / arXiv id / URL / journal name")
|
|
612
|
+
ap.add_argument("-n", type=int, default=8, help="number of results (default 8)")
|
|
613
|
+
ap.add_argument("--since", default=None, help="only papers from year YYYY onwards")
|
|
614
|
+
ap.add_argument("--topic", default=None, help="injournal: topic keywords to filter within the journal")
|
|
615
|
+
ap.add_argument("--json", action="store_true", help="structured JSON output")
|
|
616
|
+
ap.add_argument("-o", "--out", default=None, help="fetch: path to save the PDF")
|
|
617
|
+
a = ap.parse_args()
|
|
618
|
+
|
|
619
|
+
if a.cmd in ("search", "priorart", "occupancy"):
|
|
620
|
+
res = multi_search(a.query, a.n, a.since)
|
|
621
|
+
if a.json:
|
|
622
|
+
print(json.dumps(res, ensure_ascii=False, indent=2)); return
|
|
623
|
+
if not res:
|
|
624
|
+
if _net_failed():
|
|
625
|
+
print(f"Could not query the sources.\n {_net_hint()}"); return
|
|
626
|
+
print("No results. Try narrower, more focused keywords - long sentences "
|
|
627
|
+
"drag in off-topic papers."); return
|
|
628
|
+
print(f"# {len(res)} nearest papers - query: {a.query}\n")
|
|
629
|
+
for i, p in enumerate(res, 1):
|
|
630
|
+
print(fmt_paper(p, i)); print()
|
|
631
|
+
if a.cmd in ("priorart", "occupancy"):
|
|
632
|
+
lat = latest(a.query, 5) # always surface the newest work - a prior-art check must not miss it
|
|
633
|
+
if lat:
|
|
634
|
+
print(f"# Most recent related work (>={CUR_YEAR-1}) - check whether someone *just* did your angle:")
|
|
635
|
+
for p in lat: print(" " + fmt_paper(p).replace("\n", "\n "))
|
|
636
|
+
print()
|
|
637
|
+
print(OCC_TEMPLATE)
|
|
638
|
+
return
|
|
639
|
+
|
|
640
|
+
if a.cmd == "injournal":
|
|
641
|
+
jname, res = injournal(a.query, a.n, a.topic, a.since)
|
|
642
|
+
if a.json:
|
|
643
|
+
print(json.dumps(res or [], ensure_ascii=False, indent=2)); return
|
|
644
|
+
if not res:
|
|
645
|
+
print(f"Journal or its recent papers not found: {a.query} (resolved name={jname})"); return
|
|
646
|
+
print(f"# Recent papers in {jname} - study its actual conventions before submitting "
|
|
647
|
+
f"- topic: {a.topic or '(all)'}\n")
|
|
648
|
+
for i, p in enumerate(res, 1):
|
|
649
|
+
oa = "OA" if p.get("is_oa") else "closed"
|
|
650
|
+
print(fmt_paper(p, i) + f" [{oa}]"); print()
|
|
651
|
+
print('-> Use `scholarcheck fetch "<DOI>" -o out.pdf` to pull one, then read it for\n'
|
|
652
|
+
' structure, figure style, abstract format, length and how statistics are reported.')
|
|
653
|
+
print(' Note: is_oa includes free-to-read (bronze), which is not always machine-downloadable.')
|
|
654
|
+
print(' Some publishers block automated downloads; fall back to an institutional network.')
|
|
655
|
+
return
|
|
656
|
+
|
|
657
|
+
if a.cmd == "journal":
|
|
658
|
+
res = journal_lookup(a.query, a.n)
|
|
659
|
+
if a.json:
|
|
660
|
+
print(json.dumps(res or [], ensure_ascii=False, indent=2)); return
|
|
661
|
+
if not res:
|
|
662
|
+
print(f"Journal not found, or the lookup failed: {a.query}"); return
|
|
663
|
+
print(f"# Journal metrics, fetched live from OpenAlex - query: {a.query}")
|
|
664
|
+
print(" Note: if2yr = OpenAlex 2yr_mean_citedness, an impact-factor-like measure.\n"
|
|
665
|
+
" It is close to, but not, the official Clarivate IF - use JCR for that, and the\n"
|
|
666
|
+
" journal's own aims & scope page for scope.\n")
|
|
667
|
+
for p in res:
|
|
668
|
+
ifv = f"{p['if2yr']:.1f}" if p.get('if2yr') is not None else "?"
|
|
669
|
+
print(f" {p['name']} [{p.get('type','')}] {p.get('publisher') or ''}")
|
|
670
|
+
print(f" IF-proxy(2yr)={ifv} h-index={p.get('h_index','?')} works={p.get('works','?')} ISSN={p.get('issn','?')} {p.get('homepage') or ''}")
|
|
671
|
+
return
|
|
672
|
+
|
|
673
|
+
if a.cmd == "latest":
|
|
674
|
+
res = latest(a.query, a.n, a.since)
|
|
675
|
+
if a.json:
|
|
676
|
+
print(json.dumps(res, ensure_ascii=False, indent=2)); return
|
|
677
|
+
if not res:
|
|
678
|
+
if _net_failed():
|
|
679
|
+
print(f"Could not query the sources.\n {_net_hint()}"); return
|
|
680
|
+
print(f"No matches since {a.since or CUR_YEAR-1}. Try narrower keywords, "
|
|
681
|
+
"or relax --since."); return
|
|
682
|
+
print(f"# Recent related work (>={a.since or CUR_YEAR-1}, newest first) - query: {a.query}\n")
|
|
683
|
+
for i, p in enumerate(res, 1):
|
|
684
|
+
print(fmt_paper(p, i)); print()
|
|
685
|
+
return
|
|
686
|
+
|
|
687
|
+
if a.cmd == "verify":
|
|
688
|
+
exact = resolve_identifier(a.query)
|
|
689
|
+
if exact:
|
|
690
|
+
if a.json:
|
|
691
|
+
print(json.dumps(exact, ensure_ascii=False, indent=2)); return
|
|
692
|
+
print("MATCH (exact identifier)")
|
|
693
|
+
print(fmt_paper(exact))
|
|
694
|
+
for c in SOURCE_CONFLICTS:
|
|
695
|
+
print(f" ! sources disagree - {c}\n"
|
|
696
|
+
f" arXiv is authoritative for an arXiv id; the record above is theirs.")
|
|
697
|
+
return
|
|
698
|
+
if re.search(DOI_RE, a.query) or re.search(r"arxiv", a.query, re.I):
|
|
699
|
+
# It looked like an identifier and did not resolve - say that,
|
|
700
|
+
# rather than falling back to a title search that cannot succeed.
|
|
701
|
+
if _net_failed():
|
|
702
|
+
print(f"INCONCLUSIVE - could not query the sources: {a.query}\n {_net_hint()}")
|
|
703
|
+
else:
|
|
704
|
+
print(f"NOT FOUND - no record with this identifier: {a.query}\n"
|
|
705
|
+
f" Check the DOI or arXiv id; if it is correct, the work may be too "
|
|
706
|
+
f"new to be indexed.")
|
|
707
|
+
return
|
|
708
|
+
cands = search_openalex(a.query, 3) or []
|
|
709
|
+
if not cands or _match_ratio(a.query, cands[0]["title"]) < 0.6:
|
|
710
|
+
cands = cands + (search_s2(a.query, 3) or [])
|
|
711
|
+
if a.json:
|
|
712
|
+
print(json.dumps(cands, ensure_ascii=False, indent=2)); return
|
|
713
|
+
if not cands:
|
|
714
|
+
# Never call something fake when we simply could not reach the APIs.
|
|
715
|
+
if _net_failed():
|
|
716
|
+
print(f"INCONCLUSIVE - could not query the sources, so nothing can be said "
|
|
717
|
+
f"about: {a.query}\n {_net_hint()}"); return
|
|
718
|
+
print(f"NOT FOUND in any of the four sources -> this citation is very likely "
|
|
719
|
+
f"hallucinated: {a.query}"); return
|
|
720
|
+
best = max(cands, key=lambda p: _match_ratio(a.query, p.get("title", "")))
|
|
721
|
+
r = _match_ratio(a.query, best.get("title", ""))
|
|
722
|
+
verdict = ("MATCH (high confidence)" if r >= 0.75 else
|
|
723
|
+
"PARTIAL MATCH - confirm by hand that this is the same paper" if r >= 0.45 else
|
|
724
|
+
"NO MATCH -> very likely hallucinated")
|
|
725
|
+
print(f"{verdict} [query term coverage = {r:.0%}]")
|
|
726
|
+
print(fmt_paper(best))
|
|
727
|
+
if r < 0.75:
|
|
728
|
+
others = [p for p in cands if p is not best][:3]
|
|
729
|
+
if others:
|
|
730
|
+
print("\nOther candidates:"); [print(fmt_paper(p)) for p in others]
|
|
731
|
+
return
|
|
732
|
+
|
|
733
|
+
if a.cmd == "bibtex":
|
|
734
|
+
b = bibtex(a.query)
|
|
735
|
+
if isinstance(b, tuple) and b and b[0] == "INCONCLUSIVE":
|
|
736
|
+
_, cand, r = b
|
|
737
|
+
print(f"INCONCLUSIVE - a primary source could not be reached, so no entry is emitted "
|
|
738
|
+
f"for: {a.query}")
|
|
739
|
+
print(f" {_net_hint()}")
|
|
740
|
+
if cand:
|
|
741
|
+
print(f" (the partial search's best candidate was {r:.0%} coverage - not enough "
|
|
742
|
+
f"to stand on while sources are down)")
|
|
743
|
+
elif isinstance(b, tuple) and b and b[0] == "WEAK":
|
|
744
|
+
_, cand, r = b
|
|
745
|
+
if cand:
|
|
746
|
+
print(f"No confident match (best term coverage only {r:.0%}). Refusing to emit a "
|
|
747
|
+
f"possibly wrong entry. Closest candidate:")
|
|
748
|
+
print(fmt_paper(cand))
|
|
749
|
+
print('-> If that is the paper, re-run with its DOI: scholarcheck bibtex "<DOI>".\n'
|
|
750
|
+
' Otherwise give a more exact title.')
|
|
751
|
+
else:
|
|
752
|
+
if _net_failed():
|
|
753
|
+
print(f"Could not query the sources.\n {_net_hint()}")
|
|
754
|
+
else:
|
|
755
|
+
print(f"No candidates resolved - check the spelling: {a.query}")
|
|
756
|
+
elif b:
|
|
757
|
+
print(b)
|
|
758
|
+
else:
|
|
759
|
+
if _net_failed():
|
|
760
|
+
print(f"Could not query the sources.\n {_net_hint()}")
|
|
761
|
+
else:
|
|
762
|
+
print(f"Could not resolve an entry - check the spelling or DOI: {a.query}")
|
|
763
|
+
return
|
|
764
|
+
|
|
765
|
+
if a.cmd == "fetch":
|
|
766
|
+
url, label, stem = resolve_pdf(a.query)
|
|
767
|
+
if not url:
|
|
768
|
+
print(f"❌ {label}"); return
|
|
769
|
+
path = a.out or os.path.join(PDFDIR, (stem or "paper") + ".pdf")
|
|
770
|
+
if download_pdf(url, path):
|
|
771
|
+
print(f"Downloaded [{label}] -> {path}\n"
|
|
772
|
+
f" Next: read it and check the claim against the full text.")
|
|
773
|
+
else:
|
|
774
|
+
print(f"Download failed (flaky network, non-PDF response, or paywalled): {url}\n"
|
|
775
|
+
f" Retry by hand: curl -sL '{url}' -o paper.pdf")
|
|
776
|
+
return
|
|
777
|
+
|
|
778
|
+
if a.cmd == "citedby":
|
|
779
|
+
oid = resolve_openalex_id(a.query)
|
|
780
|
+
if not oid:
|
|
781
|
+
print(f"Could not resolve the target paper (try a title or DOI): {a.query}"); return
|
|
782
|
+
res = citedby_openalex(oid, a.n)
|
|
783
|
+
if a.json:
|
|
784
|
+
print(json.dumps(res or [], ensure_ascii=False, indent=2)); return
|
|
785
|
+
if not res:
|
|
786
|
+
print(f"(OpenAlex {oid}) No citing works recorded yet, or the lookup failed."); return
|
|
787
|
+
print(f"# {len(res)} papers citing {oid} (most cited first) - check whether your\n# extension is already covered:\n")
|
|
788
|
+
for i, p in enumerate(res, 1):
|
|
789
|
+
print(fmt_paper(p, i)); print()
|
|
790
|
+
return
|
|
791
|
+
|
|
792
|
+
def main():
|
|
793
|
+
"""CLI entry point. Wraps the real one so the usage nudge cannot change
|
|
794
|
+
the exit status or swallow an exception."""
|
|
795
|
+
try:
|
|
796
|
+
return _main()
|
|
797
|
+
finally:
|
|
798
|
+
from ._nudge import record_run
|
|
799
|
+
record_run()
|
|
800
|
+
|
|
801
|
+
|
|
802
|
+
if __name__ == "__main__":
|
|
803
|
+
main()
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scholarcheck
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Verify citations against real metadata - stop hallucinated references. Zero dependencies.
|
|
5
|
+
Author: Guo Cheng
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/GuoCheng24/scholarcheck
|
|
8
|
+
Project-URL: Issues, https://github.com/GuoCheng24/scholarcheck/issues
|
|
9
|
+
Keywords: citations,bibtex,openalex,crossref,arxiv,literature-review,hallucination,research-tools,doi
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# scholarcheck
|
|
22
|
+
|
|
23
|
+
[](https://github.com/GuoCheng24/scholarcheck/actions/workflows/test.yml) [](https://www.python.org/) [](LICENSE)
|
|
24
|
+
|
|
25
|
+
**Stop hallucinated citations.** Verify any reference against real metadata — from the command line, with zero dependencies.
|
|
26
|
+
|
|
27
|
+
<p align="center">
|
|
28
|
+
<img src="docs/three-states.png" width="100%">
|
|
29
|
+
</p>
|
|
30
|
+
|
|
31
|
+
<sub>The figure above is generated by <a href="docs/three-states_figure.py">docs/three-states_figure.py</a> — <code>pip install git+https://github.com/GuoCheng24/sciglyph</code> and run it to reproduce <code>docs/three-states.png</code> byte for byte.</sub>
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
Language models invent plausible-looking papers: right-sounding title, plausible authors, a DOI that resolves to nothing. `scholarcheck` answers one question honestly — **does this paper actually exist?** — by querying OpenAlex, Semantic Scholar, Crossref and arXiv directly.
|
|
35
|
+
|
|
36
|
+
```console
|
|
37
|
+
$ scholarcheck verify "Deep Residual Learning for Image Recognition"
|
|
38
|
+
MATCH (high confidence) [query term coverage = 100%]
|
|
39
|
+
Deep Residual Learning for Image Recognition (2016, conference-paper; cited=226875) doi:10.1109/cvpr.2016.90
|
|
40
|
+
Kaiming He, Xiangyu Zhang, Shaoqing Ren et al.
|
|
41
|
+
|
|
42
|
+
$ scholarcheck verify "Quantum Topological Radiomics for Zebra Diagnosis in Martian Cohorts"
|
|
43
|
+
NOT FOUND in any of the four sources -> this citation is very likely hallucinated
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Why not just ask an AI assistant?
|
|
47
|
+
|
|
48
|
+
Because an assistant answers from memory, and memory is exactly what fails here. Three design choices make this different:
|
|
49
|
+
|
|
50
|
+
**1. It says "I could not check" instead of "it is fake."**
|
|
51
|
+
A verifier that reports a network outage as *hallucinated* is worse than no verifier. `scholarcheck` tracks every failed request and distinguishes the two:
|
|
52
|
+
|
|
53
|
+
```console
|
|
54
|
+
$ scholarcheck verify "Attention Is All You Need" # with the network down
|
|
55
|
+
INCONCLUSIVE - could not query the sources, so nothing can be said about: Attention Is All You Need
|
|
56
|
+
Could not reach: api.openalex.org: curl: (7) Connection refused
|
|
57
|
+
(no proxy set; if your network needs one, set SCHOLARCHECK_PROXY)
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
It also knows which sources matter: Semantic Scholar rate-limits aggressively without an API key, so its failure never turns a real answer into "inconclusive" — only the primary sources do.
|
|
61
|
+
|
|
62
|
+
**2. It refuses to guess.**
|
|
63
|
+
Ask for BibTeX from a slightly-wrong title and most tools hand back the nearest hit. Silently citing the *wrong* paper is worse than citing none, so a weak match returns the candidate and stops:
|
|
64
|
+
|
|
65
|
+
```console
|
|
66
|
+
$ scholarcheck bibtex "Deep Residual Learning for Image Recognition in Medicine"
|
|
67
|
+
No confident match (best term coverage only 62%). Refusing to emit a possibly wrong entry.
|
|
68
|
+
Closest candidate:
|
|
69
|
+
Deep Residual Learning for Image Recognition (2016, CVPR) doi:10.1109/CVPR.2016.90
|
|
70
|
+
-> If that is the paper, re-run with its DOI: scholarcheck bibtex "<DOI>".
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The same refusal applies when the sources themselves are unavailable, which is
|
|
74
|
+
when a wrong entry is most likely — the "best" match would then be whichever
|
|
75
|
+
paper happened to be reachable:
|
|
76
|
+
|
|
77
|
+
```console
|
|
78
|
+
$ scholarcheck bibtex "Deep Residual Learning for Image Recognition in Medicine"
|
|
79
|
+
INCONCLUSIVE - a primary source could not be reached, so no entry is emitted for: ...
|
|
80
|
+
Could not reach: api.openalex.org: HTTP 429
|
|
81
|
+
(the partial search's best candidate was 50% coverage - not enough to stand on
|
|
82
|
+
while sources are down)
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
**3. An identifier is resolved, not searched.**
|
|
86
|
+
`verify "arXiv:1906.08253"` looks the identifier up directly. Feeding it to a
|
|
87
|
+
title matcher would return whatever paper happens to share those digits and
|
|
88
|
+
then score it as a mismatch — which reads as *"this citation is fake"* when the
|
|
89
|
+
truth is that the query was never looked up properly.
|
|
90
|
+
|
|
91
|
+
**4. Recency is a separate command, on purpose.**
|
|
92
|
+
Relevance ranking systematically favours highly-cited older work, which is exactly wrong when you are checking whether someone *just* published your idea. `latest` filters by recency as well as relevance.
|
|
93
|
+
|
|
94
|
+
## Install
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
pip install git+https://github.com/GuoCheng24/scholarcheck
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
Or clone and `pip install -e .` if you would rather read the source first — it is
|
|
101
|
+
one file.
|
|
102
|
+
|
|
103
|
+
**No dependencies.** Standard library plus `curl`. Nothing to break, nothing to
|
|
104
|
+
audit, and nothing that needs an API key: every source it queries is open.
|
|
105
|
+
|
|
106
|
+
<sub>Not on PyPI yet, so the git URL above is the install line that works today.
|
|
107
|
+
When it lands, `pip install scholarcheck` will too.</sub>
|
|
108
|
+
|
|
109
|
+
## Commands
|
|
110
|
+
|
|
111
|
+
| | |
|
|
112
|
+
|---|---|
|
|
113
|
+
| `verify "<title/DOI/arXiv id>"` | Is this citation real? An identifier resolves exactly; a title is matched by term coverage |
|
|
114
|
+
| `bibtex "<DOI/title>"` | A BibTeX entry — refuses to guess on a weak match |
|
|
115
|
+
| `search "<keywords>"` | Multi-source search, re-ranked by term overlap |
|
|
116
|
+
| `latest "<keywords>"` | Recent work only — relevance **and** recency |
|
|
117
|
+
| `priorart "<claim>"` | Nearest N real papers for a claim, plus a checklist for judging whether it is already taken |
|
|
118
|
+
| `citedby "<DOI/title>"` | What cited this paper — has someone already extended it? |
|
|
119
|
+
| `journal "<name>"` | Live journal metrics, instead of quoting an impact factor from memory |
|
|
120
|
+
| `injournal "<name>"` | Recent papers from one journal, to study its actual conventions |
|
|
121
|
+
| `fetch "<DOI/arXiv id>"` | Download the open-access PDF so a claim can be checked in full text |
|
|
122
|
+
|
|
123
|
+
Add `--json` to any command for structured output, `-n` for the number of results, `--since YYYY` to bound the year.
|
|
124
|
+
|
|
125
|
+
## Use as a library
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from scholarcheck import verify_citation, get_bibtex, NET_ERRORS
|
|
129
|
+
|
|
130
|
+
paper, confidence = verify_citation("Attention Is All You Need")
|
|
131
|
+
if paper is None and NET_ERRORS:
|
|
132
|
+
... # could not check — not evidence of anything
|
|
133
|
+
elif confidence >= 0.75:
|
|
134
|
+
print(get_bibtex(paper["doi"]))
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## Configuration
|
|
138
|
+
|
|
139
|
+
All optional:
|
|
140
|
+
|
|
141
|
+
| variable | effect |
|
|
142
|
+
|---|---|
|
|
143
|
+
| `SCHOLARCHECK_MAILTO` | your email — joins OpenAlex's polite pool, giving better rate limits |
|
|
144
|
+
| `SCHOLARCHECK_S2KEY` | Semantic Scholar API key (free) — avoids the frequent 429s |
|
|
145
|
+
| `SCHOLARCHECK_PROXY` | e.g. `socks5h://127.0.0.1:1080`; default is a direct connection |
|
|
146
|
+
|
|
147
|
+
Proxy behaviour is decided **solely** by `SCHOLARCHECK_PROXY`. Inherited `http_proxy` / `all_proxy` variables are stripped before each request, so the tool behaves the same on every machine.
|
|
148
|
+
|
|
149
|
+
## What it can and cannot tell you
|
|
150
|
+
|
|
151
|
+
**A match confirms the paper exists — not that the metadata you have is right.**
|
|
152
|
+
Bibliographic databases often hold several records for one work: a preprint, a
|
|
153
|
+
conference version, a publisher deposit. `verify` returns whichever record
|
|
154
|
+
matched best, so the year and venue you see may belong to a different record
|
|
155
|
+
than the one you meant to cite. Check them; the DOI is the reliable part.
|
|
156
|
+
|
|
157
|
+
**"NOT FOUND" is strong evidence, not proof.** Very new work, non-English
|
|
158
|
+
venues and some book chapters are indexed poorly. When it matters, run
|
|
159
|
+
`search` with looser keywords before concluding a reference is invented.
|
|
160
|
+
|
|
161
|
+
## Notes from real use
|
|
162
|
+
|
|
163
|
+
- **Feed focused keywords, not whole sentences.** A long claim drags in off-topic papers; two or three precise terms work far better.
|
|
164
|
+
- **`search` favours highly-cited older work.** That is what relevance ranking does. Use `latest` when the question is "has this been done recently?"
|
|
165
|
+
- **A title-only judgement is not a prior-art check.** For the closest candidates, `fetch` the PDF and read it.
|
|
166
|
+
|
|
167
|
+
## License
|
|
168
|
+
|
|
169
|
+
MIT © Guo Cheng
|
|
170
|
+
|
|
171
|
+
## 关于那行 star 提示
|
|
172
|
+
|
|
173
|
+
跑命令时,`scholarcheck` 会在**第 5 次和第 25 次**往 stderr 写一行,提一句这个仓库在哪。**一辈子只有这两次**,此外再不出声。
|
|
174
|
+
|
|
175
|
+
它不会出现在:管道或重定向里(stderr 不是终端就直接返回,连计数文件都不建)、CI 环境里(`CI` / `GITHUB_ACTIONS`)。它写的是 stderr 而非 stdout,所以不会污染你的数据输出;它包在 `try/finally` 里且吞掉自身所有异常,**不会改变退出码,也不会影响结果**。
|
|
176
|
+
|
|
177
|
+
永久关掉:
|
|
178
|
+
|
|
179
|
+
```bash
|
|
180
|
+
export SCHOLARCHECK_NO_NUDGE=1
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
计数存在 `$XDG_STATE_HOME/scholarcheck/usage.json`(默认 `~/.local/state/scholarcheck/usage.json`),删掉即重置。
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
scholarcheck/__init__.py,sha256=UCsINEupUkR93_aSX9H-F0T2cb5_47VVemK7owlisgA,1545
|
|
2
|
+
scholarcheck/_nudge.py,sha256=FDpOdxkAObhX2_otyRIDstAxxWup_0tRV7MnqV_m2jU,1897
|
|
3
|
+
scholarcheck/cli.py,sha256=r3gcTLYxmNGA8IBs2Ckkvc5jBxDszmYHo0yz0E9v0ns,38033
|
|
4
|
+
scholarcheck-0.1.0.dist-info/licenses/LICENSE,sha256=7Ujw8dU7FULnBCYz_jKV-VweojL93VUnpQnbHwo6IGk,1066
|
|
5
|
+
scholarcheck-0.1.0.dist-info/METADATA,sha256=aPx6TH4YiZY5ULO6p3r248Ya3mJrnKgLbAdE7ICSJjE,9140
|
|
6
|
+
scholarcheck-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
7
|
+
scholarcheck-0.1.0.dist-info/entry_points.txt,sha256=niNXVcH0hBfYjSSJcmpSPYr1cWmSsPmz4eUM2wM0FNg,55
|
|
8
|
+
scholarcheck-0.1.0.dist-info/top_level.txt,sha256=SCEMDuN-Vm-QlG85dKk3QZ8Q_JEg5JTsZzd0q5bdJv0,13
|
|
9
|
+
scholarcheck-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Guo Cheng
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
scholarcheck
|