ref-verify 1.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ref_verify/__init__.py +5 -0
- ref_verify/abstract_lookup.py +226 -0
- ref_verify/claim_check.py +876 -0
- ref_verify/cli.py +162 -0
- ref_verify/crossref.py +89 -0
- ref_verify/doi_check.py +191 -0
- ref_verify/models.py +89 -0
- ref_verify/numeric_claim.py +425 -0
- ref_verify/pubmed.py +156 -0
- ref_verify/semantic_scholar.py +80 -0
- ref_verify-1.1.2.dist-info/METADATA +250 -0
- ref_verify-1.1.2.dist-info/RECORD +16 -0
- ref_verify-1.1.2.dist-info/WHEEL +5 -0
- ref_verify-1.1.2.dist-info/entry_points.txt +2 -0
- ref_verify-1.1.2.dist-info/licenses/LICENSE +21 -0
- ref_verify-1.1.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
_NUMBER_PATTERN = r"\d+(?:,\d{3})*(?:\.\d+)?"
|
|
8
|
+
_UNIT_PATTERN = (
|
|
9
|
+
r"(?:mg/ml|mg\/ml|g\/l|per\s+cent|percent|cycles?|patients?|subjects?|"
|
|
10
|
+
r"samples?|devices?|°c|degc|khz|mhz|mv|kv|ma|mg|ml|kg|mm|cm|nm|hz|"
|
|
11
|
+
r"%|v|a|c|g|l|m)"
|
|
12
|
+
)
|
|
13
|
+
_MEASUREMENT_PATTERN = re.compile(
|
|
14
|
+
rf"(?<![\d,])(?P<value>{_NUMBER_PATTERN})\s*(?P<unit>{_UNIT_PATTERN})(?=$|\W)",
|
|
15
|
+
re.IGNORECASE,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
_STOPWORDS = {
|
|
19
|
+
"a",
|
|
20
|
+
"an",
|
|
21
|
+
"and",
|
|
22
|
+
"as",
|
|
23
|
+
"at",
|
|
24
|
+
"by",
|
|
25
|
+
"for",
|
|
26
|
+
"in",
|
|
27
|
+
"of",
|
|
28
|
+
"on",
|
|
29
|
+
"the",
|
|
30
|
+
"to",
|
|
31
|
+
"was",
|
|
32
|
+
"were",
|
|
33
|
+
"with",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
_NON_SUBJECT_TERMS = {
|
|
37
|
+
"above",
|
|
38
|
+
"at",
|
|
39
|
+
"below",
|
|
40
|
+
"equal",
|
|
41
|
+
"exceed",
|
|
42
|
+
"exceeded",
|
|
43
|
+
"exceeding",
|
|
44
|
+
"exceeds",
|
|
45
|
+
"greater",
|
|
46
|
+
"higher",
|
|
47
|
+
"least",
|
|
48
|
+
"less",
|
|
49
|
+
"lower",
|
|
50
|
+
"more",
|
|
51
|
+
"most",
|
|
52
|
+
"no",
|
|
53
|
+
"not",
|
|
54
|
+
"than",
|
|
55
|
+
"under",
|
|
56
|
+
"up",
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
_UNSUPPORTED_FRAME_PATTERNS = (
|
|
60
|
+
r"\baccording to\b",
|
|
61
|
+
r"\bwhether\b",
|
|
62
|
+
r"\bnot true that\b",
|
|
63
|
+
r"\bfalse that\b",
|
|
64
|
+
r"\b(?:did|do|does|is|are|was|were|has|have|had) not\b",
|
|
65
|
+
r"\b(?:fail|fails|failed) to\b",
|
|
66
|
+
r"\bwithout\b",
|
|
67
|
+
r"\bnever\b",
|
|
68
|
+
r"\bno\b.*\b(?:observed|found|shown|showed|reported|measured|demonstrated|had)\b",
|
|
69
|
+
r"\bno (?:sample|samples|specimen|specimens|device|devices|case|cases)\b",
|
|
70
|
+
r"\bnone of (?:the )?(?:sample|samples|specimen|specimens|device|devices|case|cases)\b",
|
|
71
|
+
r"\bnone (?:show|shows|showed|had|has|have|observed|found|reported|reached|met|demonstrated)\b",
|
|
72
|
+
r"\b(?:previous|prior|earlier) (?:work|study|studies|research)\b",
|
|
73
|
+
r"\b(?:may|might|would|should)\b",
|
|
74
|
+
r"\b(?:appear|appears|appeared|appearing) to\b",
|
|
75
|
+
r"\b(?:seem|seems|seemed|seeming) to\b",
|
|
76
|
+
r"\b(?:the )?(?:paper|article|study|work) (?:report|reports|reported|reporting)\b",
|
|
77
|
+
r"\b(?:the )?authors? (?:report|reports|reported|found|finds|observed|observes|noted|notes|claim|claims)\b",
|
|
78
|
+
r"\breportedly\b",
|
|
79
|
+
r"\b(?:claim|claims|claimed|claiming)\b",
|
|
80
|
+
r"\b(?:suggest|suggests|suggested|suggesting)\b",
|
|
81
|
+
r"\b(?:indicate|indicates|indicated|indicating)\b",
|
|
82
|
+
r"\b(?:imply|implies|implied|implying)\b",
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
_SCOPE_OR_COMPARATIVE_SUFFIX_TERMS = {
|
|
86
|
+
"across",
|
|
87
|
+
"after",
|
|
88
|
+
"among",
|
|
89
|
+
"at",
|
|
90
|
+
"average",
|
|
91
|
+
"averaged",
|
|
92
|
+
"averages",
|
|
93
|
+
"before",
|
|
94
|
+
"during",
|
|
95
|
+
"except",
|
|
96
|
+
"following",
|
|
97
|
+
"for",
|
|
98
|
+
"from",
|
|
99
|
+
"in",
|
|
100
|
+
"longer",
|
|
101
|
+
"max",
|
|
102
|
+
"maximum",
|
|
103
|
+
"mean",
|
|
104
|
+
"median",
|
|
105
|
+
"min",
|
|
106
|
+
"minimum",
|
|
107
|
+
"more",
|
|
108
|
+
"only",
|
|
109
|
+
"then",
|
|
110
|
+
"typical",
|
|
111
|
+
"typically",
|
|
112
|
+
"under",
|
|
113
|
+
"unless",
|
|
114
|
+
"until",
|
|
115
|
+
"when",
|
|
116
|
+
"while",
|
|
117
|
+
"within",
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
_UNIT_TERMS = {
|
|
121
|
+
"a",
|
|
122
|
+
"c",
|
|
123
|
+
"cycle",
|
|
124
|
+
"cycles",
|
|
125
|
+
"degc",
|
|
126
|
+
"device",
|
|
127
|
+
"devices",
|
|
128
|
+
"g",
|
|
129
|
+
"hz",
|
|
130
|
+
"kg",
|
|
131
|
+
"khz",
|
|
132
|
+
"kv",
|
|
133
|
+
"l",
|
|
134
|
+
"m",
|
|
135
|
+
"ma",
|
|
136
|
+
"mg",
|
|
137
|
+
"mgml",
|
|
138
|
+
"mhz",
|
|
139
|
+
"ml",
|
|
140
|
+
"mm",
|
|
141
|
+
"mv",
|
|
142
|
+
"nm",
|
|
143
|
+
"patient",
|
|
144
|
+
"patients",
|
|
145
|
+
"per",
|
|
146
|
+
"percent",
|
|
147
|
+
"sample",
|
|
148
|
+
"samples",
|
|
149
|
+
"subject",
|
|
150
|
+
"subjects",
|
|
151
|
+
"v",
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
@dataclass(frozen=True)
|
|
156
|
+
class NumericClaimResult:
|
|
157
|
+
status: str
|
|
158
|
+
reason: str
|
|
159
|
+
evidence: str
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
@dataclass(frozen=True)
|
|
163
|
+
class NumericExpression:
|
|
164
|
+
value: float
|
|
165
|
+
unit: str
|
|
166
|
+
comparator: str
|
|
167
|
+
subject_terms: set[str]
|
|
168
|
+
evidence: str
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def check_numeric_claim_support(abstract: str, claim: str) -> NumericClaimResult:
|
|
172
|
+
claim_expression = _extract_claim_expression(claim)
|
|
173
|
+
if claim_expression is None:
|
|
174
|
+
return NumericClaimResult(
|
|
175
|
+
status="NOT_NUMERIC",
|
|
176
|
+
reason="The claim does not contain a supported numeric expression.",
|
|
177
|
+
evidence="",
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
related_evidence = ""
|
|
181
|
+
for clause in _clauses(abstract):
|
|
182
|
+
if _has_unsupported_frame(clause):
|
|
183
|
+
related_evidence = related_evidence or clause
|
|
184
|
+
continue
|
|
185
|
+
evidence_expressions = _extract_evidence_expressions(clause)
|
|
186
|
+
if not evidence_expressions:
|
|
187
|
+
continue
|
|
188
|
+
if _subject_terms_match(claim_expression.subject_terms, clause):
|
|
189
|
+
related_evidence = related_evidence or clause
|
|
190
|
+
for evidence_expression in evidence_expressions:
|
|
191
|
+
if not _units_match(claim_expression.unit, evidence_expression.unit):
|
|
192
|
+
continue
|
|
193
|
+
if not _subject_terms_match(claim_expression.subject_terms, clause):
|
|
194
|
+
continue
|
|
195
|
+
if _evidence_entails_claim(
|
|
196
|
+
evidence_expression.value,
|
|
197
|
+
evidence_expression.comparator,
|
|
198
|
+
claim_expression.value,
|
|
199
|
+
claim_expression.comparator,
|
|
200
|
+
):
|
|
201
|
+
return NumericClaimResult(
|
|
202
|
+
status="SUPPORTED",
|
|
203
|
+
reason="The abstract explicitly reports a matching numeric claim.",
|
|
204
|
+
evidence=clause,
|
|
205
|
+
)
|
|
206
|
+
related_evidence = related_evidence or clause
|
|
207
|
+
|
|
208
|
+
return NumericClaimResult(
|
|
209
|
+
status="PARTIAL",
|
|
210
|
+
reason="The abstract contains numeric evidence, but not a clear subject-bound match.",
|
|
211
|
+
evidence=related_evidence or _best_numeric_clause(abstract),
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _extract_claim_expression(claim: str) -> NumericExpression | None:
|
|
216
|
+
match = _MEASUREMENT_PATTERN.search(claim)
|
|
217
|
+
if not match:
|
|
218
|
+
return None
|
|
219
|
+
return NumericExpression(
|
|
220
|
+
value=_parse_value(match.group("value")),
|
|
221
|
+
unit=_normalize_unit(match.group("unit")),
|
|
222
|
+
comparator=_claim_comparator(claim[: match.start()]),
|
|
223
|
+
subject_terms=_subject_terms(claim[: match.start()]),
|
|
224
|
+
evidence=claim,
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _extract_evidence_expressions(clause: str) -> list[NumericExpression]:
|
|
229
|
+
expressions: list[NumericExpression] = []
|
|
230
|
+
for match in _MEASUREMENT_PATTERN.finditer(clause):
|
|
231
|
+
expressions.append(
|
|
232
|
+
NumericExpression(
|
|
233
|
+
value=_parse_value(match.group("value")),
|
|
234
|
+
unit=_normalize_unit(match.group("unit")),
|
|
235
|
+
comparator=_evidence_comparator(
|
|
236
|
+
clause[: match.start()],
|
|
237
|
+
clause[match.end() :],
|
|
238
|
+
),
|
|
239
|
+
subject_terms=_subject_terms(clause[: match.start()]),
|
|
240
|
+
evidence=clause,
|
|
241
|
+
)
|
|
242
|
+
)
|
|
243
|
+
return expressions
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _clauses(value: str) -> list[str]:
|
|
247
|
+
clauses: list[str] = []
|
|
248
|
+
for sentence in re.split(r"(?<=[.!?])\s+|[.;:]\s+", value.strip()):
|
|
249
|
+
sentence = sentence.strip()
|
|
250
|
+
if not sentence:
|
|
251
|
+
continue
|
|
252
|
+
parts = [
|
|
253
|
+
part.strip()
|
|
254
|
+
for part in re.split(r",?\s+\b(?:and|but|while|whereas)\b\s+", sentence)
|
|
255
|
+
if part.strip()
|
|
256
|
+
]
|
|
257
|
+
for part in parts:
|
|
258
|
+
clauses.extend(_split_numeric_comma_clauses(part))
|
|
259
|
+
return clauses
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _split_numeric_comma_clauses(value: str) -> list[str]:
|
|
263
|
+
parts = [part.strip() for part in value.split(",") if part.strip()]
|
|
264
|
+
if len(parts) <= 1:
|
|
265
|
+
return [value.strip()]
|
|
266
|
+
if parts[0].lower().split()[:1] in (["after"], ["before"], ["under"], ["in"]):
|
|
267
|
+
return [value.strip()]
|
|
268
|
+
if sum(1 for part in parts if _MEASUREMENT_PATTERN.search(part)) >= 2:
|
|
269
|
+
return parts
|
|
270
|
+
return [value.strip()]
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _best_numeric_clause(value: str) -> str:
|
|
274
|
+
for clause in _clauses(value):
|
|
275
|
+
if _MEASUREMENT_PATTERN.search(clause):
|
|
276
|
+
return clause
|
|
277
|
+
return value.strip()
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _subject_terms_match(subject_terms: set[str], clause: str) -> bool:
|
|
281
|
+
if not subject_terms:
|
|
282
|
+
return False
|
|
283
|
+
clause_terms = {_stem(token) for token in _tokens(clause)}
|
|
284
|
+
return subject_terms <= clause_terms
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _subject_terms(value: str) -> set[str]:
|
|
288
|
+
return {
|
|
289
|
+
_stem(token)
|
|
290
|
+
for token in _tokens(value)
|
|
291
|
+
if _stem(token) not in _NON_SUBJECT_TERMS and _stem(token) not in _UNIT_TERMS
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _tokens(value: str) -> list[str]:
|
|
296
|
+
return [
|
|
297
|
+
token
|
|
298
|
+
for token in re.findall(r"[a-zA-Z°]+", value.lower().replace("/", ""))
|
|
299
|
+
if token not in _STOPWORDS
|
|
300
|
+
]
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _stem(token: str) -> str:
|
|
304
|
+
if token.endswith("s") and len(token) > 3:
|
|
305
|
+
return token[:-1]
|
|
306
|
+
return token
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _parse_value(value: str) -> float:
|
|
310
|
+
return float(value.replace(",", ""))
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _normalize_unit(value: str) -> str:
|
|
314
|
+
normalized = value.lower().replace(" ", "").replace("/", "")
|
|
315
|
+
if normalized in {"%", "percent"}:
|
|
316
|
+
return "%"
|
|
317
|
+
if normalized in {"°c", "degc", "c"}:
|
|
318
|
+
return "c"
|
|
319
|
+
if normalized in {"cycle", "cycles"}:
|
|
320
|
+
return "cycle"
|
|
321
|
+
if normalized in {"patient", "patients", "subject", "subjects"}:
|
|
322
|
+
return "person"
|
|
323
|
+
if normalized in {"sample", "samples", "device", "devices"}:
|
|
324
|
+
return normalized.rstrip("s")
|
|
325
|
+
return normalized.rstrip("s")
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def _units_match(left: str, right: str) -> bool:
|
|
329
|
+
return left == right
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _claim_comparator(prefix: str) -> str:
|
|
333
|
+
normalized = prefix.lower()
|
|
334
|
+
if ">=" in normalized or "≥" in normalized:
|
|
335
|
+
return "gte"
|
|
336
|
+
if "<=" in normalized or "≤" in normalized:
|
|
337
|
+
return "lte"
|
|
338
|
+
if ">" in normalized:
|
|
339
|
+
return "gt"
|
|
340
|
+
if "<" in normalized:
|
|
341
|
+
return "lt"
|
|
342
|
+
if re.search(r"\b(?:more|greater|higher) than or equal to\b", normalized):
|
|
343
|
+
return "gte"
|
|
344
|
+
if re.search(r"\b(?:less|lower|fewer) than or equal to\b", normalized):
|
|
345
|
+
return "lte"
|
|
346
|
+
if re.search(r"\b(?:at least|not less than)\b", normalized):
|
|
347
|
+
return "gte"
|
|
348
|
+
if re.search(r"\b(?:at most|no more than)\b", normalized):
|
|
349
|
+
return "lte"
|
|
350
|
+
if re.search(r"\b(?:below|under|less than)\b", normalized):
|
|
351
|
+
return "lt"
|
|
352
|
+
if re.search(r"\b(?:above|over|greater than|more than|exceeded|exceeds)\b", normalized):
|
|
353
|
+
return "gt"
|
|
354
|
+
return "exact"
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def _evidence_comparator(prefix: str, suffix: str = "") -> str:
|
|
358
|
+
normalized = prefix.lower().rstrip()
|
|
359
|
+
normalized_suffix = suffix.lower().lstrip()
|
|
360
|
+
if normalized.endswith((">=", "≥")):
|
|
361
|
+
return "gte"
|
|
362
|
+
if normalized.endswith(("<=", "≤")):
|
|
363
|
+
return "lte"
|
|
364
|
+
if normalized.endswith(">"):
|
|
365
|
+
return "gt"
|
|
366
|
+
if normalized.endswith("<"):
|
|
367
|
+
return "lt"
|
|
368
|
+
if re.search(r"\b(?:at least|not less than)\s*$", normalized):
|
|
369
|
+
return "gte"
|
|
370
|
+
if re.search(r"\b(?:at most|no more than)\s*$", normalized):
|
|
371
|
+
return "lte"
|
|
372
|
+
if re.search(r"\b(?:below|under|less than)\s*$", normalized):
|
|
373
|
+
return "lt"
|
|
374
|
+
if re.search(r"\b(?:above|over|greater than|more than|exceeded|exceeds|reached|survived|maintained)\s*$", normalized):
|
|
375
|
+
return "exact"
|
|
376
|
+
if re.search(r"\bup to\s*$", normalized):
|
|
377
|
+
return "up_to"
|
|
378
|
+
if re.search(r"^\s*(?:or\s+)?(?:more|greater|higher|min|minimum)\b", normalized_suffix):
|
|
379
|
+
return "gte"
|
|
380
|
+
if re.search(r"^\s*(?:or\s+)?(?:less|fewer|lower|max|maximum)\b", normalized_suffix):
|
|
381
|
+
return "lte"
|
|
382
|
+
return "exact"
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _has_unsupported_frame(value: str) -> bool:
|
|
386
|
+
normalized = " ".join(re.findall(r"[a-zA-Z0-9]+", value.lower()))
|
|
387
|
+
if any(re.search(pattern, normalized) for pattern in _UNSUPPORTED_FRAME_PATTERNS):
|
|
388
|
+
return True
|
|
389
|
+
for match in _MEASUREMENT_PATTERN.finditer(value):
|
|
390
|
+
prefix_tokens = [
|
|
391
|
+
token
|
|
392
|
+
for token in re.findall(r"[a-zA-Z]+", value[: match.start()].lower())
|
|
393
|
+
]
|
|
394
|
+
if any(
|
|
395
|
+
token in {"average", "averaged", "mean", "median", "typical", "typically"}
|
|
396
|
+
for token in prefix_tokens
|
|
397
|
+
):
|
|
398
|
+
return True
|
|
399
|
+
suffix_tokens = [
|
|
400
|
+
token
|
|
401
|
+
for token in re.findall(r"[a-zA-Z]+", value[match.end() :].lower())
|
|
402
|
+
]
|
|
403
|
+
if any(token in _SCOPE_OR_COMPARATIVE_SUFFIX_TERMS for token in suffix_tokens):
|
|
404
|
+
return True
|
|
405
|
+
prefix_tokens = re.findall(r"[a-zA-Z]+", value.lower())[:2]
|
|
406
|
+
return any(token in {"after", "before", "under", "in"} for token in prefix_tokens)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _evidence_entails_claim(
|
|
410
|
+
evidence_value: float,
|
|
411
|
+
evidence_comparator: str,
|
|
412
|
+
claim_value: float,
|
|
413
|
+
claim_comparator: str,
|
|
414
|
+
) -> bool:
|
|
415
|
+
if evidence_comparator == "up_to":
|
|
416
|
+
return claim_comparator in {"lt", "lte"} and evidence_value <= claim_value
|
|
417
|
+
if claim_comparator == "gt":
|
|
418
|
+
return evidence_value > claim_value
|
|
419
|
+
if claim_comparator == "gte":
|
|
420
|
+
return evidence_value >= claim_value
|
|
421
|
+
if claim_comparator == "lt":
|
|
422
|
+
return evidence_value < claim_value
|
|
423
|
+
if claim_comparator == "lte":
|
|
424
|
+
return evidence_value <= claim_value
|
|
425
|
+
return evidence_value == claim_value
|
ref_verify/pubmed.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
import xml.etree.ElementTree as ET
|
|
6
|
+
from typing import Any
|
|
7
|
+
from urllib.error import HTTPError
|
|
8
|
+
from urllib.parse import urlencode
|
|
9
|
+
from urllib.request import Request, urlopen
|
|
10
|
+
|
|
11
|
+
from ref_verify import __version__
|
|
12
|
+
from ref_verify.abstract_lookup import AbstractSourceError
|
|
13
|
+
from ref_verify.doi_check import normalize_doi
|
|
14
|
+
from ref_verify.models import PaperRecord
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class PubMedClient:
|
|
18
|
+
source_name = "pubmed"
|
|
19
|
+
|
|
20
|
+
def __init__(self, timeout: float = 20.0) -> None:
|
|
21
|
+
self.timeout = timeout
|
|
22
|
+
|
|
23
|
+
def fetch_record(self, doi: str) -> PaperRecord | None:
|
|
24
|
+
normalized = normalize_doi(doi)
|
|
25
|
+
pmids = self._search_pmids(normalized)
|
|
26
|
+
if not pmids:
|
|
27
|
+
raise AbstractSourceError("NOT_FOUND", "PubMed had no record for the DOI.")
|
|
28
|
+
if len(pmids) != 1:
|
|
29
|
+
raise AbstractSourceError("UNSUPPORTED", "PubMed returned multiple records for the DOI.")
|
|
30
|
+
request = Request(
|
|
31
|
+
"https://eutils.ncbi.nlm.nih.gov/entrez/eutils/efetch.fcgi?"
|
|
32
|
+
+ urlencode({"db": "pubmed", "id": pmids[0], "retmode": "xml"}),
|
|
33
|
+
headers={
|
|
34
|
+
"User-Agent": (
|
|
35
|
+
f"ref-verify/{__version__} "
|
|
36
|
+
"(+https://github.com/Moonweave-Research/ref-verify)"
|
|
37
|
+
)
|
|
38
|
+
},
|
|
39
|
+
)
|
|
40
|
+
try:
|
|
41
|
+
with urlopen(request, timeout=self.timeout) as response:
|
|
42
|
+
xml_payload = response.read().decode("utf-8")
|
|
43
|
+
except HTTPError as exc:
|
|
44
|
+
if exc.code == 404:
|
|
45
|
+
raise AbstractSourceError("NOT_FOUND", "PubMed had no record for the DOI.") from exc
|
|
46
|
+
raise
|
|
47
|
+
return parse_pubmed_article(xml_payload)
|
|
48
|
+
|
|
49
|
+
def _search_pmids(self, doi: str) -> list[str]:
|
|
50
|
+
request = Request(
|
|
51
|
+
"https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?"
|
|
52
|
+
+ urlencode(
|
|
53
|
+
{
|
|
54
|
+
"db": "pubmed",
|
|
55
|
+
"term": f"{doi}[AID]",
|
|
56
|
+
"retmode": "json",
|
|
57
|
+
"retmax": "2",
|
|
58
|
+
}
|
|
59
|
+
),
|
|
60
|
+
headers={
|
|
61
|
+
"User-Agent": (
|
|
62
|
+
f"ref-verify/{__version__} "
|
|
63
|
+
"(+https://github.com/Moonweave-Research/ref-verify)"
|
|
64
|
+
)
|
|
65
|
+
},
|
|
66
|
+
)
|
|
67
|
+
with urlopen(request, timeout=self.timeout) as response:
|
|
68
|
+
payload = json.loads(response.read().decode("utf-8"))
|
|
69
|
+
ids = payload.get("esearchresult", {}).get("idlist", [])
|
|
70
|
+
return [str(value) for value in ids if str(value).strip()]
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def parse_pubmed_article(xml_payload: str) -> PaperRecord | None:
|
|
74
|
+
root = ET.fromstring(xml_payload)
|
|
75
|
+
article = root.find(".//PubmedArticle")
|
|
76
|
+
if article is None:
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
doi = _article_doi(article)
|
|
80
|
+
abstract = _abstract_text(article)
|
|
81
|
+
if not doi or not abstract:
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
return PaperRecord(
|
|
85
|
+
doi=doi,
|
|
86
|
+
title=_text(article.find(".//ArticleTitle")) or "[title missing]",
|
|
87
|
+
authors=_authors(article),
|
|
88
|
+
year=_publication_year(article),
|
|
89
|
+
abstract=abstract,
|
|
90
|
+
source="PubMed",
|
|
91
|
+
journal=_text(article.find(".//Journal/Title")),
|
|
92
|
+
url=f"https://pubmed.ncbi.nlm.nih.gov/{_pmid(article)}/" if _pmid(article) else None,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _article_doi(article: ET.Element) -> str | None:
|
|
97
|
+
for element in article.findall(".//ArticleId"):
|
|
98
|
+
if element.attrib.get("IdType", "").lower() == "doi":
|
|
99
|
+
return _text(element)
|
|
100
|
+
for element in article.findall(".//ELocationID"):
|
|
101
|
+
if element.attrib.get("EIdType", "").lower() == "doi":
|
|
102
|
+
return _text(element)
|
|
103
|
+
return None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _abstract_text(article: ET.Element) -> str | None:
|
|
107
|
+
parts = []
|
|
108
|
+
for element in article.findall(".//Abstract/AbstractText"):
|
|
109
|
+
label = element.attrib.get("Label")
|
|
110
|
+
text = _flatten_text(element)
|
|
111
|
+
if not text:
|
|
112
|
+
continue
|
|
113
|
+
parts.append(f"{label}: {text}" if label else text)
|
|
114
|
+
if not parts:
|
|
115
|
+
return None
|
|
116
|
+
return re.sub(r"\s+", " ", " ".join(parts)).strip()
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _authors(article: ET.Element) -> list[str]:
|
|
120
|
+
names: list[str] = []
|
|
121
|
+
for author in article.findall(".//AuthorList/Author"):
|
|
122
|
+
collective = _text(author.find("CollectiveName"))
|
|
123
|
+
if collective:
|
|
124
|
+
names.append(collective)
|
|
125
|
+
continue
|
|
126
|
+
last_name = _text(author.find("LastName"))
|
|
127
|
+
if last_name:
|
|
128
|
+
names.append(last_name)
|
|
129
|
+
return names
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _publication_year(article: ET.Element) -> int | None:
|
|
133
|
+
for path in (".//ArticleDate/Year", ".//JournalIssue/PubDate/Year"):
|
|
134
|
+
value = _text(article.find(path))
|
|
135
|
+
if value and value.isdigit():
|
|
136
|
+
return int(value)
|
|
137
|
+
medline_date = _text(article.find(".//JournalIssue/PubDate/MedlineDate"))
|
|
138
|
+
if medline_date:
|
|
139
|
+
match = re.search(r"\b(19|20)\d{2}\b", medline_date)
|
|
140
|
+
if match:
|
|
141
|
+
return int(match.group(0))
|
|
142
|
+
return None
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _pmid(article: ET.Element) -> str | None:
|
|
146
|
+
return _text(article.find(".//PMID"))
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _text(element: ET.Element | None) -> str | None:
|
|
150
|
+
if element is None or element.text is None or not element.text.strip():
|
|
151
|
+
return None
|
|
152
|
+
return element.text.strip()
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _flatten_text(element: ET.Element) -> str:
|
|
156
|
+
return "".join(element.itertext()).strip()
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from typing import Any
|
|
5
|
+
from urllib.error import HTTPError
|
|
6
|
+
from urllib.parse import quote
|
|
7
|
+
from urllib.request import Request, urlopen
|
|
8
|
+
|
|
9
|
+
from ref_verify import __version__
|
|
10
|
+
from ref_verify.abstract_lookup import AbstractSourceError
|
|
11
|
+
from ref_verify.doi_check import normalize_doi
|
|
12
|
+
from ref_verify.models import PaperRecord
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class SemanticScholarClient:
|
|
16
|
+
source_name = "semantic_scholar"
|
|
17
|
+
|
|
18
|
+
def __init__(self, timeout: float = 20.0) -> None:
|
|
19
|
+
self.timeout = timeout
|
|
20
|
+
|
|
21
|
+
def fetch_record(self, doi: str) -> PaperRecord | None:
|
|
22
|
+
paper_id = quote(f"DOI:{normalize_doi(doi)}", safe=":")
|
|
23
|
+
fields = "title,authors,year,abstract,externalIds,url,venue"
|
|
24
|
+
request = Request(
|
|
25
|
+
f"https://api.semanticscholar.org/graph/v1/paper/{paper_id}?fields={fields}",
|
|
26
|
+
headers={
|
|
27
|
+
"User-Agent": (
|
|
28
|
+
f"ref-verify/{__version__} "
|
|
29
|
+
"(+https://github.com/Moonweave-Research/ref-verify)"
|
|
30
|
+
)
|
|
31
|
+
},
|
|
32
|
+
)
|
|
33
|
+
try:
|
|
34
|
+
with urlopen(request, timeout=self.timeout) as response:
|
|
35
|
+
payload = json.loads(response.read().decode("utf-8"))
|
|
36
|
+
except HTTPError as exc:
|
|
37
|
+
if exc.code == 404:
|
|
38
|
+
raise AbstractSourceError("NOT_FOUND", "Semantic Scholar had no paper for the DOI.") from exc
|
|
39
|
+
raise
|
|
40
|
+
return parse_semantic_scholar_paper(payload)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def parse_semantic_scholar_paper(payload: dict[str, Any]) -> PaperRecord | None:
|
|
44
|
+
abstract = _string_or_none(payload.get("abstract"))
|
|
45
|
+
if abstract is None:
|
|
46
|
+
return None
|
|
47
|
+
external_ids = payload.get("externalIds")
|
|
48
|
+
doi = ""
|
|
49
|
+
if isinstance(external_ids, dict):
|
|
50
|
+
doi = _string_or_none(external_ids.get("DOI")) or ""
|
|
51
|
+
if not doi:
|
|
52
|
+
return None
|
|
53
|
+
|
|
54
|
+
return PaperRecord(
|
|
55
|
+
doi=doi,
|
|
56
|
+
title=_string_or_none(payload.get("title")) or "[title missing]",
|
|
57
|
+
authors=[
|
|
58
|
+
name
|
|
59
|
+
for author in payload.get("authors", [])
|
|
60
|
+
if isinstance(author, dict)
|
|
61
|
+
if (name := _string_or_none(author.get("name")))
|
|
62
|
+
],
|
|
63
|
+
year=_int_or_none(payload.get("year")),
|
|
64
|
+
abstract=abstract,
|
|
65
|
+
source="Semantic Scholar",
|
|
66
|
+
journal=_string_or_none(payload.get("venue")),
|
|
67
|
+
url=_string_or_none(payload.get("url")),
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _string_or_none(value: Any) -> str | None:
|
|
72
|
+
if isinstance(value, str) and value.strip():
|
|
73
|
+
return value.strip()
|
|
74
|
+
return None
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _int_or_none(value: Any) -> int | None:
|
|
78
|
+
if isinstance(value, int):
|
|
79
|
+
return value
|
|
80
|
+
return None
|