ref-verify 1.3.0__tar.gz → 1.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ref_verify-1.3.0/src/ref_verify.egg-info → ref_verify-1.3.1}/PKG-INFO +15 -3
- {ref_verify-1.3.0 → ref_verify-1.3.1}/README.md +14 -2
- {ref_verify-1.3.0 → ref_verify-1.3.1}/pyproject.toml +1 -1
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/__init__.py +1 -1
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/cli.py +48 -2
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/crossref.py +47 -1
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/doi_check.py +1 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/http.py +25 -3
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/models.py +5 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/reference_parse.py +1 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/reference_resolve.py +216 -24
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/semantic_scholar.py +17 -3
- {ref_verify-1.3.0 → ref_verify-1.3.1/src/ref_verify.egg-info}/PKG-INFO +15 -3
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_abstract_sources.py +19 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_cli.py +74 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_http_cache.py +65 -1
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_reference_resolve.py +246 -2
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_report.py +1 -1
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_skill_docs.py +2 -2
- {ref_verify-1.3.0 → ref_verify-1.3.1}/LICENSE +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/setup.cfg +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/abstract_lookup.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/batch.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/cache.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/claim_check.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/numeric_claim.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/openalex.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/pubmed.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/report.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/SOURCES.txt +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/dependency_links.txt +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/entry_points.txt +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/top_level.txt +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_batch.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_benchmark_aggregate.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_claim_check.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_crossref.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_doi_check.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_numeric_claim.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_package_smoke.py +0 -0
- {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_reference_parse.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ref-verify
|
|
3
|
-
Version: 1.3.
|
|
3
|
+
Version: 1.3.1
|
|
4
4
|
Summary: Executable DOI and claim verification helpers for academic citations
|
|
5
5
|
Author: Moonweave Research
|
|
6
6
|
License-Expression: MIT
|
|
@@ -392,7 +392,8 @@ Current `check-claim` error codes:
|
|
|
392
392
|
- `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
|
|
393
393
|
- `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
|
|
394
394
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
395
|
-
- `DOI_NOT_FOUND`: CrossRef
|
|
395
|
+
- `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI, or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
|
|
396
|
+
- `DOI_NOT_IN_CROSSREF`: CrossRef has no record, but doi.org lists the DOI with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...). `verify-doi` returns `verdict: WARN`, `status: UNVERIFIED` without comparing metadata; `check-claim` tries OpenAlex, Semantic Scholar (arXiv DOIs by arXiv identifier), and PubMed for the abstract and judges the claim if one has it, otherwise returns `status: UNVERIFIABLE`, `verdict: WARN` with this code. Not a dead DOI.
|
|
396
397
|
- `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
|
|
397
398
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
398
399
|
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
@@ -419,7 +420,18 @@ romanized ones (윤 → Yoon/Yun). When a DOI is unknown to CrossRef, doi.org is
|
|
|
419
420
|
asked which agency registered it, so arXiv, Zenodo, or KISTI DOIs are not
|
|
420
421
|
reported as dead. Search results that are about the paper rather than the
|
|
421
422
|
paper itself (peer-review reports, Faculty Opinions recommendations,
|
|
422
|
-
addenda and corrections) are skipped.
|
|
423
|
+
addenda and corrections) are skipped. A citation in a style that omits the
|
|
424
|
+
article title (`J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108,
|
|
425
|
+
1175 (1957)`) is compared on journal (full name or abbreviation), volume, first
|
|
426
|
+
page or article number, year, and first author; it passes when all of them
|
|
427
|
+
agree, and otherwise the reason names each field that differs. Without a DOI,
|
|
428
|
+
when the plain search finds nothing and the citation looks title-less, a second
|
|
429
|
+
CrossRef search by first author,
|
|
430
|
+
the rest of the citation, and the cited year finds short citations such as
|
|
431
|
+
`A. G. Riess et al., Astron. J. 116, 1009 (1998).`; the same full agreement is
|
|
432
|
+
required. The first author
|
|
433
|
+
is read only from the first name in the list, so a reference that puts a
|
|
434
|
+
co-author first does not pass. The terminal output starts with a count line
|
|
423
435
|
(`19 references: 11 PASS, 2 WARN, 5 REJECT, 1 UNVERIFIED`), lists one row per
|
|
424
436
|
reference (citation key, or the start of the reference for a pasted list), puts
|
|
425
437
|
the reason under every row that is not `PASS`, and ends with a one-paragraph
|
|
@@ -381,7 +381,8 @@ Current `check-claim` error codes:
|
|
|
381
381
|
- `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
|
|
382
382
|
- `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
|
|
383
383
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
384
|
-
- `DOI_NOT_FOUND`: CrossRef
|
|
384
|
+
- `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI, or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
|
|
385
|
+
- `DOI_NOT_IN_CROSSREF`: CrossRef has no record, but doi.org lists the DOI with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...). `verify-doi` returns `verdict: WARN`, `status: UNVERIFIED` without comparing metadata; `check-claim` tries OpenAlex, Semantic Scholar (arXiv DOIs by arXiv identifier), and PubMed for the abstract and judges the claim if one has it, otherwise returns `status: UNVERIFIABLE`, `verdict: WARN` with this code. Not a dead DOI.
|
|
385
386
|
- `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
|
|
386
387
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
387
388
|
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
@@ -408,7 +409,18 @@ romanized ones (윤 → Yoon/Yun). When a DOI is unknown to CrossRef, doi.org is
|
|
|
408
409
|
asked which agency registered it, so arXiv, Zenodo, or KISTI DOIs are not
|
|
409
410
|
reported as dead. Search results that are about the paper rather than the
|
|
410
411
|
paper itself (peer-review reports, Faculty Opinions recommendations,
|
|
411
|
-
addenda and corrections) are skipped.
|
|
412
|
+
addenda and corrections) are skipped. A citation in a style that omits the
|
|
413
|
+
article title (`J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108,
|
|
414
|
+
1175 (1957)`) is compared on journal (full name or abbreviation), volume, first
|
|
415
|
+
page or article number, year, and first author; it passes when all of them
|
|
416
|
+
agree, and otherwise the reason names each field that differs. Without a DOI,
|
|
417
|
+
when the plain search finds nothing and the citation looks title-less, a second
|
|
418
|
+
CrossRef search by first author,
|
|
419
|
+
the rest of the citation, and the cited year finds short citations such as
|
|
420
|
+
`A. G. Riess et al., Astron. J. 116, 1009 (1998).`; the same full agreement is
|
|
421
|
+
required. The first author
|
|
422
|
+
is read only from the first name in the list, so a reference that puts a
|
|
423
|
+
co-author first does not pass. The terminal output starts with a count line
|
|
412
424
|
(`19 references: 11 PASS, 2 WARN, 5 REJECT, 1 UNVERIFIED`), lists one row per
|
|
413
425
|
reference (citation key, or the start of the reference for a pasted list), puts
|
|
414
426
|
the reason under every row that is not `PASS`, and ends with a one-paragraph
|
|
@@ -24,7 +24,7 @@ from ref_verify.batch import (
|
|
|
24
24
|
)
|
|
25
25
|
from ref_verify.cache import ResponseCache, default_cache
|
|
26
26
|
from ref_verify.claim_check import check_claim_support, retracted_claim_result
|
|
27
|
-
from ref_verify.crossref import CrossrefClient
|
|
27
|
+
from ref_verify.crossref import CrossrefClient, not_in_crossref_reason, other_registration_agency
|
|
28
28
|
from ref_verify.doi_check import normalize_doi, verify_doi_metadata
|
|
29
29
|
from ref_verify.models import CitationInput, ClaimSupportResult
|
|
30
30
|
from ref_verify.openalex import OpenAlexClient
|
|
@@ -185,6 +185,21 @@ def _verify_doi(args: argparse.Namespace, client: CrossrefClient) -> int:
|
|
|
185
185
|
except HTTPError as exc:
|
|
186
186
|
if not _is_not_found(exc):
|
|
187
187
|
raise
|
|
188
|
+
agency = other_registration_agency(client, lookup_doi)
|
|
189
|
+
if agency:
|
|
190
|
+
_emit(
|
|
191
|
+
{
|
|
192
|
+
"status": "UNVERIFIED",
|
|
193
|
+
"verdict": "WARN",
|
|
194
|
+
"mismatches": [],
|
|
195
|
+
"reason": not_in_crossref_reason(agency, lookup_doi),
|
|
196
|
+
"provided": provided.to_dict(),
|
|
197
|
+
"fetched": None,
|
|
198
|
+
"error_code": "DOI_NOT_IN_CROSSREF",
|
|
199
|
+
},
|
|
200
|
+
as_json=args.json,
|
|
201
|
+
)
|
|
202
|
+
return 2
|
|
188
203
|
# A dead DOI is a verdict, not a tool failure; keep the `verdict` key so
|
|
189
204
|
# callers that parse JSON do not have to special-case the error shape.
|
|
190
205
|
_emit(
|
|
@@ -365,7 +380,10 @@ def _run_claim_check(
|
|
|
365
380
|
except HTTPError as exc:
|
|
366
381
|
if not _is_not_found(exc):
|
|
367
382
|
raise
|
|
368
|
-
|
|
383
|
+
agency = other_registration_agency(client, lookup_doi)
|
|
384
|
+
if not agency:
|
|
385
|
+
return _doi_not_found_payload(claim)
|
|
386
|
+
return _not_in_crossref_claim(lookup_doi, claim, agency, selected_clients)
|
|
369
387
|
if fetched.retraction_doi:
|
|
370
388
|
# Fallback abstract sources would drop the retraction flag, so decide here.
|
|
371
389
|
result = retracted_claim_result(fetched, claim)
|
|
@@ -410,6 +428,34 @@ def _doi_not_found_payload(claim: str) -> dict:
|
|
|
410
428
|
}
|
|
411
429
|
|
|
412
430
|
|
|
431
|
+
def _not_in_crossref_claim(
|
|
432
|
+
doi: str,
|
|
433
|
+
claim: str,
|
|
434
|
+
agency: str,
|
|
435
|
+
fallback_clients: Sequence[AbstractSourceClient],
|
|
436
|
+
) -> dict:
|
|
437
|
+
# CrossRef has no record, but OpenAlex or Semantic Scholar often index arXiv and other
|
|
438
|
+
# DataCite works with their abstracts.
|
|
439
|
+
lookup_result = lookup_selected_abstract(doi, fallback_clients)
|
|
440
|
+
if lookup_result.record.abstract:
|
|
441
|
+
return _claim_payload(check_claim_support(lookup_result.record, claim), lookup_result)
|
|
442
|
+
sources = ", ".join(client.source_name for client in fallback_clients) or "no other source"
|
|
443
|
+
return {
|
|
444
|
+
"status": "UNVERIFIABLE",
|
|
445
|
+
"verdict": "WARN",
|
|
446
|
+
"reason": (
|
|
447
|
+
f"This DOI is registered with {agency}, not CrossRef, and no DOI-bound abstract was "
|
|
448
|
+
f"found ({sources} tried), so the claim was not judged; open https://doi.org/{doi}."
|
|
449
|
+
),
|
|
450
|
+
"evidence": "",
|
|
451
|
+
"paper": None,
|
|
452
|
+
"claim": claim,
|
|
453
|
+
"abstract_source": None,
|
|
454
|
+
"source_attempts": [attempt.to_dict() for attempt in lookup_result.attempts],
|
|
455
|
+
"error_code": "DOI_NOT_IN_CROSSREF",
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
|
|
413
459
|
def _row_error_payload(claim: str, exc: Exception) -> dict:
|
|
414
460
|
return {
|
|
415
461
|
"status": "UNVERIFIABLE",
|
|
@@ -46,8 +46,19 @@ class CrossrefClient:
|
|
|
46
46
|
)
|
|
47
47
|
return parse_crossref_work(payload["message"])
|
|
48
48
|
|
|
49
|
-
def search_bibliographic(
|
|
49
|
+
def search_bibliographic(
|
|
50
|
+
self,
|
|
51
|
+
query: str,
|
|
52
|
+
rows: int = 3,
|
|
53
|
+
*,
|
|
54
|
+
author: str | None = None,
|
|
55
|
+
year_range: tuple[int, int] | None = None,
|
|
56
|
+
) -> list[PaperRecord]:
|
|
50
57
|
params = {"query.bibliographic": query, "rows": str(rows)}
|
|
58
|
+
if author:
|
|
59
|
+
params["query.author"] = author
|
|
60
|
+
if year_range:
|
|
61
|
+
params["filter"] = f"from-pub-date:{year_range[0]},until-pub-date:{year_range[1]}"
|
|
51
62
|
mailto = os.environ.get("REF_VERIFY_MAILTO")
|
|
52
63
|
if mailto:
|
|
53
64
|
params["mailto"] = mailto
|
|
@@ -87,6 +98,26 @@ def _pace_search(interval: float) -> None:
|
|
|
87
98
|
_last_search_started[0] = time.monotonic()
|
|
88
99
|
|
|
89
100
|
|
|
101
|
+
def other_registration_agency(client: Any, doi: str) -> str | None:
|
|
102
|
+
# arXiv and Zenodo (DataCite), many Korean (KISTI) and Japanese (JaLC) DOIs are registered
|
|
103
|
+
# outside CrossRef, so a CrossRef 404 alone does not make them dead. Returns that agency,
|
|
104
|
+
# or None when doi.org names CrossRef, says the DOI does not exist, or cannot be reached.
|
|
105
|
+
try:
|
|
106
|
+
agency = client.registration_agency(doi)
|
|
107
|
+
except Exception:
|
|
108
|
+
return None
|
|
109
|
+
if agency and agency.casefold() != "crossref":
|
|
110
|
+
return agency
|
|
111
|
+
return None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def not_in_crossref_reason(agency: str, doi: str) -> str:
|
|
115
|
+
return (
|
|
116
|
+
f"This DOI is registered with {agency}, not CrossRef, so its title and authors were not "
|
|
117
|
+
f"compared; open https://doi.org/{doi} to confirm it is this work."
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
90
121
|
def parse_crossref_work(message: dict[str, Any]) -> PaperRecord:
|
|
91
122
|
doi = str(message.get("DOI") or "")
|
|
92
123
|
title = _clean_title(_first_string(message.get("title"))) or "[title missing]"
|
|
@@ -124,6 +155,11 @@ def parse_crossref_work(message: dict[str, Any]) -> PaperRecord:
|
|
|
124
155
|
alt_years=years[1:],
|
|
125
156
|
work_type=str(message["type"]) if message.get("type") else None,
|
|
126
157
|
is_about_other_work=_is_about_other_work(message, title),
|
|
158
|
+
journal_abbreviations=[
|
|
159
|
+
str(value).strip() for value in message.get("short-container-title") or [] if str(value).strip()
|
|
160
|
+
],
|
|
161
|
+
volume=_first_string(message.get("volume")),
|
|
162
|
+
first_page=_first_page(message),
|
|
127
163
|
)
|
|
128
164
|
|
|
129
165
|
|
|
@@ -172,6 +208,16 @@ def _is_about_other_work(message: dict[str, Any], title: str) -> bool:
|
|
|
172
208
|
return bool(_ABOUT_OTHER_WORK_TITLE.match(title))
|
|
173
209
|
|
|
174
210
|
|
|
211
|
+
def _first_page(message: dict[str, Any]) -> str | None:
|
|
212
|
+
article_number = _first_string(message.get("article-number"))
|
|
213
|
+
if article_number:
|
|
214
|
+
return article_number
|
|
215
|
+
page = _first_string(message.get("page"))
|
|
216
|
+
if not page:
|
|
217
|
+
return None
|
|
218
|
+
return re.split(r"\s*[-\u2013\u2014,]\s*", page)[0] or None
|
|
219
|
+
|
|
220
|
+
|
|
175
221
|
def _first_string(value: Any) -> str | None:
|
|
176
222
|
if isinstance(value, list) and value:
|
|
177
223
|
return str(value[0]).strip()
|
|
@@ -274,6 +274,7 @@ def _transliterate_greek_letters(value: str) -> str:
|
|
|
274
274
|
titles_match = _titles_match
|
|
275
275
|
author_matches = _author_matches
|
|
276
276
|
author_tokens = _author_tokens
|
|
277
|
+
looks_like_group_author = _looks_like_group_author
|
|
277
278
|
|
|
278
279
|
|
|
279
280
|
def title_in_text(title: str, text: str) -> bool:
|
|
@@ -3,7 +3,7 @@ from __future__ import annotations
|
|
|
3
3
|
import json
|
|
4
4
|
import time
|
|
5
5
|
from email.message import Message
|
|
6
|
-
from typing import Any, Callable
|
|
6
|
+
from typing import Any, Callable, Protocol
|
|
7
7
|
from urllib.error import HTTPError
|
|
8
8
|
from urllib.request import Request, urlopen
|
|
9
9
|
|
|
@@ -15,6 +15,29 @@ RETRY_STATUSES = frozenset({429, 500, 502, 503, 504})
|
|
|
15
15
|
MAX_BACKOFF_SECONDS = 10.0
|
|
16
16
|
|
|
17
17
|
|
|
18
|
+
class HttpBackend(Protocol):
|
|
19
|
+
# Returns the decoded body of a 2xx response. An HTTP error status raises
|
|
20
|
+
# urllib.error.HTTPError (with `code` and `headers`) so retries, Retry-After, and the
|
|
21
|
+
# 404 cache work the same for every backend; a network failure raises URLError.
|
|
22
|
+
def get(self, url: str, headers: dict[str, str], timeout: float) -> str:
|
|
23
|
+
...
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class UrllibBackend:
|
|
27
|
+
def get(self, url: str, headers: dict[str, str], timeout: float) -> str:
|
|
28
|
+
with urlopen(Request(url, headers=headers), timeout=timeout) as response:
|
|
29
|
+
return response.read().decode("utf-8")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
_backend: HttpBackend = UrllibBackend()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def set_backend(backend: HttpBackend | None) -> None:
|
|
36
|
+
# The browser build swaps in a transport that runs inside the page; None restores urllib.
|
|
37
|
+
global _backend
|
|
38
|
+
_backend = backend if backend is not None else UrllibBackend()
|
|
39
|
+
|
|
40
|
+
|
|
18
41
|
def fetch_json(
|
|
19
42
|
url: str,
|
|
20
43
|
*,
|
|
@@ -54,8 +77,7 @@ def fetch_text(
|
|
|
54
77
|
attempt = 0
|
|
55
78
|
while True:
|
|
56
79
|
try:
|
|
57
|
-
|
|
58
|
-
body = response.read().decode("utf-8")
|
|
80
|
+
body = _backend.get(url, headers, timeout)
|
|
59
81
|
except HTTPError as exc:
|
|
60
82
|
if exc.code == 404 and cache is not None:
|
|
61
83
|
cache.put(url, 404, "")
|
|
@@ -34,6 +34,11 @@ class PaperRecord:
|
|
|
34
34
|
# review, recommendation, or notice about another work rather than the work itself.
|
|
35
35
|
work_type: str | None = None
|
|
36
36
|
is_about_other_work: bool = False
|
|
37
|
+
# Where the paper sits in its journal, for citation styles that omit the article title
|
|
38
|
+
# ("Phys. Rev. 108, 1175 (1957)"); `first_page` is the article number when there is one.
|
|
39
|
+
journal_abbreviations: list[str] = field(default_factory=list)
|
|
40
|
+
volume: str | None = None
|
|
41
|
+
first_page: str | None = None
|
|
37
42
|
|
|
38
43
|
def to_dict(self) -> dict[str, Any]:
|
|
39
44
|
return asdict(self)
|
|
@@ -21,6 +21,7 @@ _SUFFIX_FORMATS: dict[str, ReferenceFormat] = {
|
|
|
21
21
|
# Old Wiley DOIs are SICIs with an angle-bracketed part ("...40:11<2004::aid-anie2004>3.0.co;2-5");
|
|
22
22
|
# a bracketed run without spaces stays in the DOI, while a DOI wrapped as <...> ends at ">".
|
|
23
23
|
_DOI_PATTERN = re.compile(r"10\.\d{4,9}/(?:[^\s\"<>{}]|<[^\s\"<>{}]*>)+", re.IGNORECASE)
|
|
24
|
+
DOI_PATTERN = _DOI_PATTERN
|
|
24
25
|
_YEAR_PATTERN = re.compile(r"\b(?:1[89]|20)\d{2}\b")
|
|
25
26
|
_PARENTHESISED_YEAR_PATTERN = re.compile(r"\(((?:1[89]|20)\d{2})[a-z]?\)")
|
|
26
27
|
# A publication year is followed by punctuation or the end ("2020;395", "2020.", "2020)"),
|
|
@@ -13,6 +13,7 @@ from ref_verify.doi_check import (
|
|
|
13
13
|
author_matches,
|
|
14
14
|
author_tokens,
|
|
15
15
|
hangul_surname_matches,
|
|
16
|
+
looks_like_group_author,
|
|
16
17
|
record_titles,
|
|
17
18
|
record_years,
|
|
18
19
|
title_in_text,
|
|
@@ -20,9 +21,10 @@ from ref_verify.doi_check import (
|
|
|
20
21
|
titles_match,
|
|
21
22
|
verify_doi_metadata,
|
|
22
23
|
)
|
|
24
|
+
from ref_verify.crossref import not_in_crossref_reason, other_registration_agency
|
|
23
25
|
from ref_verify.http import retry_after_seconds as _retry_after_seconds
|
|
24
26
|
from ref_verify.models import CitationInput, PaperRecord
|
|
25
|
-
from ref_verify.reference_parse import ReferenceEntry
|
|
27
|
+
from ref_verify.reference_parse import DOI_PATTERN, ReferenceEntry
|
|
26
28
|
|
|
27
29
|
UNMATCHED_REASON = "No matching CrossRef record was found; verify this reference manually."
|
|
28
30
|
_INSUFFICIENT_REASON = (
|
|
@@ -32,9 +34,21 @@ _INSUFFICIENT_REASON = (
|
|
|
32
34
|
_MIN_TEXT_TITLE_OVERLAP = 0.8
|
|
33
35
|
# Below this share of the CrossRef title's words, plain reference text is about another paper.
|
|
34
36
|
_MAX_SWAPPED_TITLE_OVERLAP = 0.5
|
|
35
|
-
# Reference strings start with the author list
|
|
36
|
-
#
|
|
37
|
-
|
|
37
|
+
# Reference strings start with the author list. The first author is the text before the
|
|
38
|
+
# first separator ("Pelrine R, ...", "R. Pelrine, ...", "Pelrine, R., ...", "A. G. Riess et
|
|
39
|
+
# al.", "Ronald E. Pelrine and ..."), and the family name sits within its first few words
|
|
40
|
+
# ("J. D. van der Waals").
|
|
41
|
+
_FIRST_AUTHOR_END = re.compile(r",|;|\(|\s(?:and|&)\s|\set\.?\s*al\b", re.IGNORECASE)
|
|
42
|
+
_FIRST_AUTHOR_TOKEN_WINDOW = 6
|
|
43
|
+
# Words that are neither title nor author in a citation such as
|
|
44
|
+
# "J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108, 1175 (1957)".
|
|
45
|
+
_CITATION_FILLER = {
|
|
46
|
+
"al", "and", "et", "vol", "no", "pp", "doi", "https", "http", "org", "dx", "art", "article",
|
|
47
|
+
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "sept", "oct", "nov", "dec",
|
|
48
|
+
}
|
|
49
|
+
# More words than this before the volume number are a title plus journal, not a journal alone.
|
|
50
|
+
_MAX_TITLELESS_JOURNAL_WORDS = 5
|
|
51
|
+
_JOURNAL_STOPWORDS = {"a", "an", "and", "de", "der", "des", "for", "in", "of", "on", "the", "und"}
|
|
38
52
|
_DEFAULT_RATE_LIMIT_PAUSE_SECONDS = 15.0
|
|
39
53
|
_MAX_RATE_LIMIT_PAUSE_SECONDS = 60.0
|
|
40
54
|
_sleep = time.sleep
|
|
@@ -45,7 +59,14 @@ class ReferenceClient(Protocol):
|
|
|
45
59
|
def fetch_work(self, doi: str) -> PaperRecord:
|
|
46
60
|
...
|
|
47
61
|
|
|
48
|
-
def search_bibliographic(
|
|
62
|
+
def search_bibliographic(
|
|
63
|
+
self,
|
|
64
|
+
query: str,
|
|
65
|
+
rows: int = 3,
|
|
66
|
+
*,
|
|
67
|
+
author: str | None = None,
|
|
68
|
+
year_range: tuple[int, int] | None = None,
|
|
69
|
+
) -> list[PaperRecord]:
|
|
49
70
|
...
|
|
50
71
|
|
|
51
72
|
def registration_agency(self, doi: str) -> str | None:
|
|
@@ -147,7 +168,25 @@ def _check_doi_reference(entry: ReferenceEntry, doi: str, client: ReferenceClien
|
|
|
147
168
|
if metadata.verdict == "PASS":
|
|
148
169
|
status, reason = "VERIFIED", metadata.reason
|
|
149
170
|
elif metadata.verdict == "WARN" and "metadata" in metadata.mismatches:
|
|
150
|
-
|
|
171
|
+
titleless = None if text_title else _titleless_differences(entry, fetched)
|
|
172
|
+
if titleless == []:
|
|
173
|
+
return ReferenceResult(
|
|
174
|
+
entry=entry,
|
|
175
|
+
status="VERIFIED",
|
|
176
|
+
verdict="PASS",
|
|
177
|
+
reason=(
|
|
178
|
+
"The reference has no article title; it matches CrossRef's record for this DOI on "
|
|
179
|
+
f"{_titleless_fields(fetched)}."
|
|
180
|
+
),
|
|
181
|
+
fetched=fetched,
|
|
182
|
+
)
|
|
183
|
+
if titleless:
|
|
184
|
+
status = "MISMATCH"
|
|
185
|
+
reason = (
|
|
186
|
+
"The reference has no article title, and CrossRef's record for this DOI differs in "
|
|
187
|
+
f"{'; '.join(titleless)}."
|
|
188
|
+
)
|
|
189
|
+
elif not entry.title and not text_title and _text_names_another_paper(entry.raw, fetched):
|
|
151
190
|
status = "MISMATCH"
|
|
152
191
|
reason = (
|
|
153
192
|
f"This DOI belongs to {_describe(fetched)}, which this reference does not mention; "
|
|
@@ -176,21 +215,13 @@ def _check_doi_reference(entry: ReferenceEntry, doi: str, client: ReferenceClien
|
|
|
176
215
|
|
|
177
216
|
|
|
178
217
|
def _missing_from_crossref(entry: ReferenceEntry, doi: str, client: ReferenceClient) -> ReferenceResult:
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
except Exception:
|
|
182
|
-
agency = None
|
|
183
|
-
# arXiv, Zenodo, and many Korean (KISTI) or Japanese (JaLC) DOIs are registered with
|
|
184
|
-
# another agency, so a CrossRef 404 alone does not make them dead.
|
|
185
|
-
if agency and agency.casefold() != "crossref":
|
|
218
|
+
agency = other_registration_agency(client, doi)
|
|
219
|
+
if agency:
|
|
186
220
|
return ReferenceResult(
|
|
187
221
|
entry=entry,
|
|
188
222
|
status="UNVERIFIED",
|
|
189
223
|
verdict="WARN",
|
|
190
|
-
reason=(
|
|
191
|
-
f"This DOI is registered with {agency}, not CrossRef, so its title and authors "
|
|
192
|
-
f"were not compared; open https://doi.org/{doi} to confirm it is this work."
|
|
193
|
-
),
|
|
224
|
+
reason=not_in_crossref_reason(agency, doi),
|
|
194
225
|
error_code="DOI_NOT_IN_CROSSREF",
|
|
195
226
|
)
|
|
196
227
|
return ReferenceResult(
|
|
@@ -216,7 +247,7 @@ def _insufficient_reason(provided: CitationInput, fetched: PaperRecord) -> str:
|
|
|
216
247
|
if provided.title and not provided.first_author and fetched.authors:
|
|
217
248
|
return (
|
|
218
249
|
f"The DOI and title match CrossRef, but its first author ({fetched.authors[0]}) was not "
|
|
219
|
-
"found in the reference; check the author names manually."
|
|
250
|
+
"found in the reference's first-author position; check the author names manually."
|
|
220
251
|
)
|
|
221
252
|
if provided.first_author and not provided.title:
|
|
222
253
|
return (
|
|
@@ -258,7 +289,11 @@ def _resolve_reference(entry: ReferenceEntry, client: ReferenceClient) -> Refere
|
|
|
258
289
|
# top three results ahead of the paper itself.
|
|
259
290
|
candidates = client.search_bibliographic(query, rows=5) if query else []
|
|
260
291
|
for candidate in candidates:
|
|
261
|
-
if candidate.is_about_other_work
|
|
292
|
+
if candidate.is_about_other_work:
|
|
293
|
+
continue
|
|
294
|
+
if not _candidate_matches(entry, candidate):
|
|
295
|
+
if _titleless_differences(entry, candidate) == []:
|
|
296
|
+
return _titleless_resolved(entry, candidate)
|
|
262
297
|
continue
|
|
263
298
|
if candidate.retraction_doi:
|
|
264
299
|
return _retracted(entry, candidate, resolved_doi=candidate.doi)
|
|
@@ -282,6 +317,16 @@ def _resolve_reference(entry: ReferenceEntry, client: ReferenceClient) -> Refere
|
|
|
282
317
|
resolved_doi=candidate.doi,
|
|
283
318
|
fetched=candidate,
|
|
284
319
|
)
|
|
320
|
+
structured = _structured_query(entry)
|
|
321
|
+
if structured:
|
|
322
|
+
# A short title-less citation ("A. G. Riess et al., Astron. J. 116, 1009 (1998).") can
|
|
323
|
+
# rank below unrelated records with similar numbers; a second search by first author,
|
|
324
|
+
# journal, volume, and page within the cited year finds it. Only full field agreement
|
|
325
|
+
# is accepted from it.
|
|
326
|
+
query, author, year_range = structured
|
|
327
|
+
for candidate in client.search_bibliographic(query, rows=5, author=author, year_range=year_range):
|
|
328
|
+
if not candidate.is_about_other_work and _titleless_differences(entry, candidate) == []:
|
|
329
|
+
return _titleless_resolved(entry, candidate)
|
|
285
330
|
return ReferenceResult(
|
|
286
331
|
entry=entry,
|
|
287
332
|
status="UNVERIFIED",
|
|
@@ -291,6 +336,66 @@ def _resolve_reference(entry: ReferenceEntry, client: ReferenceClient) -> Refere
|
|
|
291
336
|
)
|
|
292
337
|
|
|
293
338
|
|
|
339
|
+
def _titleless_resolved(entry: ReferenceEntry, candidate: PaperRecord) -> ReferenceResult:
|
|
340
|
+
if candidate.retraction_doi:
|
|
341
|
+
return _retracted(entry, candidate, resolved_doi=candidate.doi)
|
|
342
|
+
return ReferenceResult(
|
|
343
|
+
entry=entry,
|
|
344
|
+
status="RESOLVED",
|
|
345
|
+
verdict="PASS",
|
|
346
|
+
reason=(
|
|
347
|
+
f"Matched CrossRef record {candidate.doi} by bibliographic search; the reference has "
|
|
348
|
+
f"no article title, so it was matched on {_titleless_fields(candidate)}."
|
|
349
|
+
),
|
|
350
|
+
error_code="REFERENCE_RESOLVED",
|
|
351
|
+
resolved_doi=candidate.doi,
|
|
352
|
+
fetched=candidate,
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _structured_query(entry: ReferenceEntry) -> tuple[str, str, tuple[int, int]] | None:
|
|
357
|
+
# The first author goes to `query.author`, the rest of the citation without the year (a
|
|
358
|
+
# bare year matches far too many records) to `query.bibliographic`, and the year (or the
|
|
359
|
+
# year before, for online-first papers) to a publication-date filter.
|
|
360
|
+
if entry.title or entry.year is None or not _looks_titleless(entry.raw):
|
|
361
|
+
return None
|
|
362
|
+
parts = _FIRST_AUTHOR_END.split(DOI_PATTERN.sub(" ", entry.raw), maxsplit=1)
|
|
363
|
+
names = re.findall(r"[^\W\d_]{2,}", parts[0])
|
|
364
|
+
if len(parts) < 2 or not names:
|
|
365
|
+
return None
|
|
366
|
+
rest = re.sub(r"\bet\.?\s*al\b\.?|\(?\b(?:1[89]|20)\d{2}[a-z]?\b\)?", " ", parts[1])
|
|
367
|
+
rest = " ".join(rest.split()).strip(" .,;")
|
|
368
|
+
if not rest:
|
|
369
|
+
return None
|
|
370
|
+
return rest, names[-1], (entry.year - 1, entry.year)
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _looks_titleless(text: str) -> bool:
|
|
374
|
+
# Without a record to compare against: the words before the volume number, once the
|
|
375
|
+
# authors are set aside, are only a journal name ("A. G. Riess et al., Astron. J. 116,
|
|
376
|
+
# 1009"), not a title plus journal. Authors are the first word and any word next to an
|
|
377
|
+
# initial ("Riess" in "A. G. Riess", "Cooper" in "L. N. Cooper", "Tang" in "Tang, C. W.").
|
|
378
|
+
text = re.sub(r"\bet\.?\s*al\b|\(?\b(?:1[89]|20)\d{2}[a-z]?\b\)?", " ", DOI_PATTERN.sub(" ", text))
|
|
379
|
+
tokens = re.findall(r"[^\W_]+", text)
|
|
380
|
+
numbers = [index for index, token in enumerate(tokens) if token.isdigit()]
|
|
381
|
+
# Volume and page (or article number) must both follow the journal.
|
|
382
|
+
if len(numbers) < 2:
|
|
383
|
+
return False
|
|
384
|
+
before = tokens[: numbers[0]]
|
|
385
|
+
initials = {index for index, token in enumerate(before) if token.isalpha() and token.isupper() and len(token) <= 2}
|
|
386
|
+
words = [
|
|
387
|
+
token
|
|
388
|
+
for index, token in enumerate(before)
|
|
389
|
+
if index not in initials
|
|
390
|
+
and index != 0
|
|
391
|
+
and not ({index - 1, index + 1} & initials)
|
|
392
|
+
and len(token) >= 3
|
|
393
|
+
and token.isalpha()
|
|
394
|
+
and token.casefold() not in _JOURNAL_STOPWORDS | _CITATION_FILLER
|
|
395
|
+
]
|
|
396
|
+
return len(words) <= _MAX_TITLELESS_JOURNAL_WORDS
|
|
397
|
+
|
|
398
|
+
|
|
294
399
|
def _bibliographic_query(entry: ReferenceEntry) -> str:
|
|
295
400
|
if entry.title:
|
|
296
401
|
parts = (entry.title, entry.first_author, entry.year, entry.journal)
|
|
@@ -326,11 +431,98 @@ def _first_author_matches(entry: ReferenceEntry, candidate: PaperRecord) -> bool
|
|
|
326
431
|
def _first_author_in_text(text: str, record: PaperRecord) -> str | None:
|
|
327
432
|
if not text or not record.authors:
|
|
328
433
|
return None
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
434
|
+
first_author = record.authors[0]
|
|
435
|
+
if hangul_surname_matches(text, first_author):
|
|
436
|
+
return first_author
|
|
437
|
+
family = author_tokens(first_author)
|
|
438
|
+
if not family:
|
|
439
|
+
return None
|
|
440
|
+
if looks_like_group_author(family):
|
|
441
|
+
# A group author ("Writing Group for the ... Investigators") opens the reference whole.
|
|
442
|
+
group = [token for token in family if token != "the"]
|
|
443
|
+
opening = [token for token in author_tokens(text) if token != "the"][: len(group)]
|
|
444
|
+
return first_author if opening == group else None
|
|
445
|
+
# Only the first author's own name counts, so "Perlmutter S, Riess AG" does not pass for
|
|
446
|
+
# a paper whose first author is Riess.
|
|
447
|
+
first_name = _FIRST_AUTHOR_END.split(text, maxsplit=1)[0]
|
|
448
|
+
if family[-1] in author_tokens(first_name)[:_FIRST_AUTHOR_TOKEN_WINDOW]:
|
|
449
|
+
return first_author
|
|
450
|
+
return None
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _titleless_differences(entry: ReferenceEntry, record: PaperRecord) -> list[str] | None:
|
|
454
|
+
# A citation without an article title ("Phys. Rev. 108, 1175 (1957)") is compared on the
|
|
455
|
+
# fields it does carry. None means it carries a title (words that are not authors,
|
|
456
|
+
# journal, or numbers) or the record has no volume or page to compare; otherwise the list
|
|
457
|
+
# names the fields that disagree, and an empty list means every field agrees.
|
|
458
|
+
if entry.title or not (record.volume or record.first_page):
|
|
459
|
+
return None
|
|
460
|
+
tokens = _plain_tokens(DOI_PATTERN.sub(" ", entry.raw))
|
|
461
|
+
journal_run = next(
|
|
462
|
+
(
|
|
463
|
+
run
|
|
464
|
+
for journal in (record.journal, *record.journal_abbreviations)
|
|
465
|
+
if journal and (run := _journal_run(journal, tokens))
|
|
466
|
+
),
|
|
467
|
+
None,
|
|
468
|
+
)
|
|
469
|
+
author_words = {token for author in record.authors for token in author_tokens(author)}
|
|
470
|
+
leftover = [
|
|
471
|
+
token
|
|
472
|
+
for index, token in enumerate(tokens)
|
|
473
|
+
if not (journal_run and index in journal_run)
|
|
474
|
+
and token not in author_words
|
|
475
|
+
and token not in _CITATION_FILLER
|
|
476
|
+
and not token.isdigit()
|
|
477
|
+
and len(token) >= 3
|
|
478
|
+
]
|
|
479
|
+
if len(leftover) >= 3:
|
|
480
|
+
return None
|
|
481
|
+
differences = []
|
|
482
|
+
if not journal_run:
|
|
483
|
+
differences.append(f"journal (CrossRef: {record.journal or 'none listed'})")
|
|
484
|
+
if record.volume and record.volume.casefold() not in tokens:
|
|
485
|
+
differences.append(f"volume (CrossRef: {record.volume})")
|
|
486
|
+
if record.first_page and record.first_page.casefold() not in tokens:
|
|
487
|
+
differences.append(f"first page (CrossRef: {record.first_page})")
|
|
488
|
+
years = record_years(record)
|
|
489
|
+
if entry.year is None or entry.year not in years:
|
|
490
|
+
reference = f"reference: {entry.year}; " if entry.year is not None else ""
|
|
491
|
+
crossref_years = "/".join(str(year) for year in years) or "none listed"
|
|
492
|
+
differences.append(f"year ({reference}CrossRef: {crossref_years})")
|
|
493
|
+
if _first_author_in_text(entry.raw, record) is None:
|
|
494
|
+
differences.append(f"first author (CrossRef: {record.authors[0] if record.authors else 'none listed'})")
|
|
495
|
+
return differences
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def _titleless_fields(record: PaperRecord) -> str:
|
|
499
|
+
fields = ["journal"]
|
|
500
|
+
if record.volume:
|
|
501
|
+
fields.append("volume")
|
|
502
|
+
if record.first_page:
|
|
503
|
+
fields.append("first page")
|
|
504
|
+
fields.extend(["year", "first author"])
|
|
505
|
+
return ", ".join(fields[:-1]) + f", and {fields[-1]}"
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def _plain_tokens(text: str) -> list[str]:
|
|
509
|
+
folded = "".join(
|
|
510
|
+
char for char in unicodedata.normalize("NFKD", text.casefold()) if not unicodedata.combining(char)
|
|
511
|
+
)
|
|
512
|
+
return re.findall(r"[a-z0-9]+", folded)
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def _journal_run(journal: str, tokens: list[str]) -> set[int] | None:
|
|
516
|
+
# Full name or a standard abbreviation: "Phys. Rev. Lett." matches "Physical Review
|
|
517
|
+
# Letters" word by word, each abbreviated word being the start of the full one.
|
|
518
|
+
words = [word for word in _plain_tokens(journal) if word not in _JOURNAL_STOPWORDS]
|
|
519
|
+
if not words:
|
|
520
|
+
return None
|
|
521
|
+
positions = [index for index, token in enumerate(tokens) if token not in _JOURNAL_STOPWORDS]
|
|
522
|
+
for start in range(len(positions) - len(words) + 1):
|
|
523
|
+
window = positions[start : start + len(words)]
|
|
524
|
+
if all(words[offset].startswith(tokens[index]) for offset, index in enumerate(window)):
|
|
525
|
+
return set(range(window[0], window[-1] + 1))
|
|
334
526
|
return None
|
|
335
527
|
|
|
336
528
|
|