ref-verify 1.3.0__tar.gz → 1.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {ref_verify-1.3.0/src/ref_verify.egg-info → ref_verify-1.3.1}/PKG-INFO +15 -3
  2. {ref_verify-1.3.0 → ref_verify-1.3.1}/README.md +14 -2
  3. {ref_verify-1.3.0 → ref_verify-1.3.1}/pyproject.toml +1 -1
  4. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/__init__.py +1 -1
  5. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/cli.py +48 -2
  6. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/crossref.py +47 -1
  7. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/doi_check.py +1 -0
  8. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/http.py +25 -3
  9. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/models.py +5 -0
  10. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/reference_parse.py +1 -0
  11. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/reference_resolve.py +216 -24
  12. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/semantic_scholar.py +17 -3
  13. {ref_verify-1.3.0 → ref_verify-1.3.1/src/ref_verify.egg-info}/PKG-INFO +15 -3
  14. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_abstract_sources.py +19 -0
  15. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_cli.py +74 -0
  16. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_http_cache.py +65 -1
  17. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_reference_resolve.py +246 -2
  18. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_report.py +1 -1
  19. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_skill_docs.py +2 -2
  20. {ref_verify-1.3.0 → ref_verify-1.3.1}/LICENSE +0 -0
  21. {ref_verify-1.3.0 → ref_verify-1.3.1}/setup.cfg +0 -0
  22. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/abstract_lookup.py +0 -0
  23. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/batch.py +0 -0
  24. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/cache.py +0 -0
  25. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/claim_check.py +0 -0
  26. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/numeric_claim.py +0 -0
  27. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/openalex.py +0 -0
  28. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/pubmed.py +0 -0
  29. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify/report.py +0 -0
  30. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/SOURCES.txt +0 -0
  31. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/dependency_links.txt +0 -0
  32. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/entry_points.txt +0 -0
  33. {ref_verify-1.3.0 → ref_verify-1.3.1}/src/ref_verify.egg-info/top_level.txt +0 -0
  34. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_batch.py +0 -0
  35. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_benchmark_aggregate.py +0 -0
  36. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_claim_check.py +0 -0
  37. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_crossref.py +0 -0
  38. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_doi_check.py +0 -0
  39. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_numeric_claim.py +0 -0
  40. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_package_smoke.py +0 -0
  41. {ref_verify-1.3.0 → ref_verify-1.3.1}/tests/test_reference_parse.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ref-verify
3
- Version: 1.3.0
3
+ Version: 1.3.1
4
4
  Summary: Executable DOI and claim verification helpers for academic citations
5
5
  Author: Moonweave Research
6
6
  License-Expression: MIT
@@ -392,7 +392,8 @@ Current `check-claim` error codes:
392
392
  - `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
393
393
  - `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
394
394
  - `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
395
- - `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
395
+ - `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI, or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
396
+ - `DOI_NOT_IN_CROSSREF`: CrossRef has no record, but doi.org lists the DOI with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...). `verify-doi` returns `verdict: WARN`, `status: UNVERIFIED` without comparing metadata; `check-claim` tries OpenAlex, Semantic Scholar (arXiv DOIs by arXiv identifier), and PubMed for the abstract and judges the claim if one has it, otherwise returns `status: UNVERIFIABLE`, `verdict: WARN` with this code. Not a dead DOI.
396
397
  - `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
397
398
  - `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
398
399
  - `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
@@ -419,7 +420,18 @@ romanized ones (윤 → Yoon/Yun). When a DOI is unknown to CrossRef, doi.org is
419
420
  asked which agency registered it, so arXiv, Zenodo, or KISTI DOIs are not
420
421
  reported as dead. Search results that are about the paper rather than the
421
422
  paper itself (peer-review reports, Faculty Opinions recommendations,
422
- addenda and corrections) are skipped. The terminal output starts with a count line
423
+ addenda and corrections) are skipped. A citation in a style that omits the
424
+ article title (`J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108,
425
+ 1175 (1957)`) is compared on journal (full name or abbreviation), volume, first
426
+ page or article number, year, and first author; it passes when all of them
427
+ agree, and otherwise the reason names each field that differs. Without a DOI,
428
+ when the plain search finds nothing and the citation looks title-less, a second
429
+ CrossRef search by first author,
430
+ the rest of the citation, and the cited year finds short citations such as
431
+ `A. G. Riess et al., Astron. J. 116, 1009 (1998).`; the same full agreement is
432
+ required. The first author
433
+ is read only from the first name in the list, so a reference that puts a
434
+ co-author first does not pass. The terminal output starts with a count line
423
435
  (`19 references: 11 PASS, 2 WARN, 5 REJECT, 1 UNVERIFIED`), lists one row per
424
436
  reference (citation key, or the start of the reference for a pasted list), puts
425
437
  the reason under every row that is not `PASS`, and ends with a one-paragraph
@@ -381,7 +381,8 @@ Current `check-claim` error codes:
381
381
  - `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
382
382
  - `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
383
383
  - `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
384
- - `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
384
+ - `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI, or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
385
+ - `DOI_NOT_IN_CROSSREF`: CrossRef has no record, but doi.org lists the DOI with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...). `verify-doi` returns `verdict: WARN`, `status: UNVERIFIED` without comparing metadata; `check-claim` tries OpenAlex, Semantic Scholar (arXiv DOIs by arXiv identifier), and PubMed for the abstract and judges the claim if one has it, otherwise returns `status: UNVERIFIABLE`, `verdict: WARN` with this code. Not a dead DOI.
385
386
  - `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
386
387
  - `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
387
388
  - `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
@@ -408,7 +409,18 @@ romanized ones (윤 → Yoon/Yun). When a DOI is unknown to CrossRef, doi.org is
408
409
  asked which agency registered it, so arXiv, Zenodo, or KISTI DOIs are not
409
410
  reported as dead. Search results that are about the paper rather than the
410
411
  paper itself (peer-review reports, Faculty Opinions recommendations,
411
- addenda and corrections) are skipped. The terminal output starts with a count line
412
+ addenda and corrections) are skipped. A citation in a style that omits the
413
+ article title (`J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108,
414
+ 1175 (1957)`) is compared on journal (full name or abbreviation), volume, first
415
+ page or article number, year, and first author; it passes when all of them
416
+ agree, and otherwise the reason names each field that differs. Without a DOI,
417
+ when the plain search finds nothing and the citation looks title-less, a second
418
+ CrossRef search by first author,
419
+ the rest of the citation, and the cited year finds short citations such as
420
+ `A. G. Riess et al., Astron. J. 116, 1009 (1998).`; the same full agreement is
421
+ required. The first author
422
+ is read only from the first name in the list, so a reference that puts a
423
+ co-author first does not pass. The terminal output starts with a count line
412
424
  (`19 references: 11 PASS, 2 WARN, 5 REJECT, 1 UNVERIFIED`), lists one row per
413
425
  reference (citation key, or the start of the reference for a pasted list), puts
414
426
  the reason under every row that is not `PASS`, and ends with a one-paragraph
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "ref-verify"
7
- version = "1.3.0"
7
+ version = "1.3.1"
8
8
  description = "Executable DOI and claim verification helpers for academic citations"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -2,4 +2,4 @@
2
2
 
3
3
  __all__ = ["__version__"]
4
4
 
5
- __version__ = "1.3.0"
5
+ __version__ = "1.3.1"
@@ -24,7 +24,7 @@ from ref_verify.batch import (
24
24
  )
25
25
  from ref_verify.cache import ResponseCache, default_cache
26
26
  from ref_verify.claim_check import check_claim_support, retracted_claim_result
27
- from ref_verify.crossref import CrossrefClient
27
+ from ref_verify.crossref import CrossrefClient, not_in_crossref_reason, other_registration_agency
28
28
  from ref_verify.doi_check import normalize_doi, verify_doi_metadata
29
29
  from ref_verify.models import CitationInput, ClaimSupportResult
30
30
  from ref_verify.openalex import OpenAlexClient
@@ -185,6 +185,21 @@ def _verify_doi(args: argparse.Namespace, client: CrossrefClient) -> int:
185
185
  except HTTPError as exc:
186
186
  if not _is_not_found(exc):
187
187
  raise
188
+ agency = other_registration_agency(client, lookup_doi)
189
+ if agency:
190
+ _emit(
191
+ {
192
+ "status": "UNVERIFIED",
193
+ "verdict": "WARN",
194
+ "mismatches": [],
195
+ "reason": not_in_crossref_reason(agency, lookup_doi),
196
+ "provided": provided.to_dict(),
197
+ "fetched": None,
198
+ "error_code": "DOI_NOT_IN_CROSSREF",
199
+ },
200
+ as_json=args.json,
201
+ )
202
+ return 2
188
203
  # A dead DOI is a verdict, not a tool failure; keep the `verdict` key so
189
204
  # callers that parse JSON do not have to special-case the error shape.
190
205
  _emit(
@@ -365,7 +380,10 @@ def _run_claim_check(
365
380
  except HTTPError as exc:
366
381
  if not _is_not_found(exc):
367
382
  raise
368
- return _doi_not_found_payload(claim)
383
+ agency = other_registration_agency(client, lookup_doi)
384
+ if not agency:
385
+ return _doi_not_found_payload(claim)
386
+ return _not_in_crossref_claim(lookup_doi, claim, agency, selected_clients)
369
387
  if fetched.retraction_doi:
370
388
  # Fallback abstract sources would drop the retraction flag, so decide here.
371
389
  result = retracted_claim_result(fetched, claim)
@@ -410,6 +428,34 @@ def _doi_not_found_payload(claim: str) -> dict:
410
428
  }
411
429
 
412
430
 
431
+ def _not_in_crossref_claim(
432
+ doi: str,
433
+ claim: str,
434
+ agency: str,
435
+ fallback_clients: Sequence[AbstractSourceClient],
436
+ ) -> dict:
437
+ # CrossRef has no record, but OpenAlex or Semantic Scholar often index arXiv and other
438
+ # DataCite works with their abstracts.
439
+ lookup_result = lookup_selected_abstract(doi, fallback_clients)
440
+ if lookup_result.record.abstract:
441
+ return _claim_payload(check_claim_support(lookup_result.record, claim), lookup_result)
442
+ sources = ", ".join(client.source_name for client in fallback_clients) or "no other source"
443
+ return {
444
+ "status": "UNVERIFIABLE",
445
+ "verdict": "WARN",
446
+ "reason": (
447
+ f"This DOI is registered with {agency}, not CrossRef, and no DOI-bound abstract was "
448
+ f"found ({sources} tried), so the claim was not judged; open https://doi.org/{doi}."
449
+ ),
450
+ "evidence": "",
451
+ "paper": None,
452
+ "claim": claim,
453
+ "abstract_source": None,
454
+ "source_attempts": [attempt.to_dict() for attempt in lookup_result.attempts],
455
+ "error_code": "DOI_NOT_IN_CROSSREF",
456
+ }
457
+
458
+
413
459
  def _row_error_payload(claim: str, exc: Exception) -> dict:
414
460
  return {
415
461
  "status": "UNVERIFIABLE",
@@ -46,8 +46,19 @@ class CrossrefClient:
46
46
  )
47
47
  return parse_crossref_work(payload["message"])
48
48
 
49
- def search_bibliographic(self, query: str, rows: int = 3) -> list[PaperRecord]:
49
+ def search_bibliographic(
50
+ self,
51
+ query: str,
52
+ rows: int = 3,
53
+ *,
54
+ author: str | None = None,
55
+ year_range: tuple[int, int] | None = None,
56
+ ) -> list[PaperRecord]:
50
57
  params = {"query.bibliographic": query, "rows": str(rows)}
58
+ if author:
59
+ params["query.author"] = author
60
+ if year_range:
61
+ params["filter"] = f"from-pub-date:{year_range[0]},until-pub-date:{year_range[1]}"
51
62
  mailto = os.environ.get("REF_VERIFY_MAILTO")
52
63
  if mailto:
53
64
  params["mailto"] = mailto
@@ -87,6 +98,26 @@ def _pace_search(interval: float) -> None:
87
98
  _last_search_started[0] = time.monotonic()
88
99
 
89
100
 
101
+ def other_registration_agency(client: Any, doi: str) -> str | None:
102
+ # arXiv and Zenodo (DataCite), many Korean (KISTI) and Japanese (JaLC) DOIs are registered
103
+ # outside CrossRef, so a CrossRef 404 alone does not make them dead. Returns that agency,
104
+ # or None when doi.org names CrossRef, says the DOI does not exist, or cannot be reached.
105
+ try:
106
+ agency = client.registration_agency(doi)
107
+ except Exception:
108
+ return None
109
+ if agency and agency.casefold() != "crossref":
110
+ return agency
111
+ return None
112
+
113
+
114
+ def not_in_crossref_reason(agency: str, doi: str) -> str:
115
+ return (
116
+ f"This DOI is registered with {agency}, not CrossRef, so its title and authors were not "
117
+ f"compared; open https://doi.org/{doi} to confirm it is this work."
118
+ )
119
+
120
+
90
121
  def parse_crossref_work(message: dict[str, Any]) -> PaperRecord:
91
122
  doi = str(message.get("DOI") or "")
92
123
  title = _clean_title(_first_string(message.get("title"))) or "[title missing]"
@@ -124,6 +155,11 @@ def parse_crossref_work(message: dict[str, Any]) -> PaperRecord:
124
155
  alt_years=years[1:],
125
156
  work_type=str(message["type"]) if message.get("type") else None,
126
157
  is_about_other_work=_is_about_other_work(message, title),
158
+ journal_abbreviations=[
159
+ str(value).strip() for value in message.get("short-container-title") or [] if str(value).strip()
160
+ ],
161
+ volume=_first_string(message.get("volume")),
162
+ first_page=_first_page(message),
127
163
  )
128
164
 
129
165
 
@@ -172,6 +208,16 @@ def _is_about_other_work(message: dict[str, Any], title: str) -> bool:
172
208
  return bool(_ABOUT_OTHER_WORK_TITLE.match(title))
173
209
 
174
210
 
211
+ def _first_page(message: dict[str, Any]) -> str | None:
212
+ article_number = _first_string(message.get("article-number"))
213
+ if article_number:
214
+ return article_number
215
+ page = _first_string(message.get("page"))
216
+ if not page:
217
+ return None
218
+ return re.split(r"\s*[-\u2013\u2014,]\s*", page)[0] or None
219
+
220
+
175
221
  def _first_string(value: Any) -> str | None:
176
222
  if isinstance(value, list) and value:
177
223
  return str(value[0]).strip()
@@ -274,6 +274,7 @@ def _transliterate_greek_letters(value: str) -> str:
274
274
  titles_match = _titles_match
275
275
  author_matches = _author_matches
276
276
  author_tokens = _author_tokens
277
+ looks_like_group_author = _looks_like_group_author
277
278
 
278
279
 
279
280
  def title_in_text(title: str, text: str) -> bool:
@@ -3,7 +3,7 @@ from __future__ import annotations
3
3
  import json
4
4
  import time
5
5
  from email.message import Message
6
- from typing import Any, Callable
6
+ from typing import Any, Callable, Protocol
7
7
  from urllib.error import HTTPError
8
8
  from urllib.request import Request, urlopen
9
9
 
@@ -15,6 +15,29 @@ RETRY_STATUSES = frozenset({429, 500, 502, 503, 504})
15
15
  MAX_BACKOFF_SECONDS = 10.0
16
16
 
17
17
 
18
+ class HttpBackend(Protocol):
19
+ # Returns the decoded body of a 2xx response. An HTTP error status raises
20
+ # urllib.error.HTTPError (with `code` and `headers`) so retries, Retry-After, and the
21
+ # 404 cache work the same for every backend; a network failure raises URLError.
22
+ def get(self, url: str, headers: dict[str, str], timeout: float) -> str:
23
+ ...
24
+
25
+
26
+ class UrllibBackend:
27
+ def get(self, url: str, headers: dict[str, str], timeout: float) -> str:
28
+ with urlopen(Request(url, headers=headers), timeout=timeout) as response:
29
+ return response.read().decode("utf-8")
30
+
31
+
32
+ _backend: HttpBackend = UrllibBackend()
33
+
34
+
35
+ def set_backend(backend: HttpBackend | None) -> None:
36
+ # The browser build swaps in a transport that runs inside the page; None restores urllib.
37
+ global _backend
38
+ _backend = backend if backend is not None else UrllibBackend()
39
+
40
+
18
41
  def fetch_json(
19
42
  url: str,
20
43
  *,
@@ -54,8 +77,7 @@ def fetch_text(
54
77
  attempt = 0
55
78
  while True:
56
79
  try:
57
- with urlopen(Request(url, headers=headers), timeout=timeout) as response:
58
- body = response.read().decode("utf-8")
80
+ body = _backend.get(url, headers, timeout)
59
81
  except HTTPError as exc:
60
82
  if exc.code == 404 and cache is not None:
61
83
  cache.put(url, 404, "")
@@ -34,6 +34,11 @@ class PaperRecord:
34
34
  # review, recommendation, or notice about another work rather than the work itself.
35
35
  work_type: str | None = None
36
36
  is_about_other_work: bool = False
37
+ # Where the paper sits in its journal, for citation styles that omit the article title
38
+ # ("Phys. Rev. 108, 1175 (1957)"); `first_page` is the article number when there is one.
39
+ journal_abbreviations: list[str] = field(default_factory=list)
40
+ volume: str | None = None
41
+ first_page: str | None = None
37
42
 
38
43
  def to_dict(self) -> dict[str, Any]:
39
44
  return asdict(self)
@@ -21,6 +21,7 @@ _SUFFIX_FORMATS: dict[str, ReferenceFormat] = {
21
21
  # Old Wiley DOIs are SICIs with an angle-bracketed part ("...40:11<2004::aid-anie2004>3.0.co;2-5");
22
22
  # a bracketed run without spaces stays in the DOI, while a DOI wrapped as <...> ends at ">".
23
23
  _DOI_PATTERN = re.compile(r"10\.\d{4,9}/(?:[^\s\"<>{}]|<[^\s\"<>{}]*>)+", re.IGNORECASE)
24
+ DOI_PATTERN = _DOI_PATTERN
24
25
  _YEAR_PATTERN = re.compile(r"\b(?:1[89]|20)\d{2}\b")
25
26
  _PARENTHESISED_YEAR_PATTERN = re.compile(r"\(((?:1[89]|20)\d{2})[a-z]?\)")
26
27
  # A publication year is followed by punctuation or the end ("2020;395", "2020.", "2020)"),
@@ -13,6 +13,7 @@ from ref_verify.doi_check import (
13
13
  author_matches,
14
14
  author_tokens,
15
15
  hangul_surname_matches,
16
+ looks_like_group_author,
16
17
  record_titles,
17
18
  record_years,
18
19
  title_in_text,
@@ -20,9 +21,10 @@ from ref_verify.doi_check import (
20
21
  titles_match,
21
22
  verify_doi_metadata,
22
23
  )
24
+ from ref_verify.crossref import not_in_crossref_reason, other_registration_agency
23
25
  from ref_verify.http import retry_after_seconds as _retry_after_seconds
24
26
  from ref_verify.models import CitationInput, PaperRecord
25
- from ref_verify.reference_parse import ReferenceEntry
27
+ from ref_verify.reference_parse import DOI_PATTERN, ReferenceEntry
26
28
 
27
29
  UNMATCHED_REASON = "No matching CrossRef record was found; verify this reference manually."
28
30
  _INSUFFICIENT_REASON = (
@@ -32,9 +34,21 @@ _INSUFFICIENT_REASON = (
32
34
  _MIN_TEXT_TITLE_OVERLAP = 0.8
33
35
  # Below this share of the CrossRef title's words, plain reference text is about another paper.
34
36
  _MAX_SWAPPED_TITLE_OVERLAP = 0.5
35
- # Reference strings start with the author list, so the first author's family name should
36
- # appear among the first few name tokens ("Pelrine R", "R. Pelrine", "Ronald E. Pelrine").
37
- _FIRST_AUTHOR_TOKEN_WINDOW = 4
37
+ # Reference strings start with the author list. The first author is the text before the
38
+ # first separator ("Pelrine R, ...", "R. Pelrine, ...", "Pelrine, R., ...", "A. G. Riess et
39
+ # al.", "Ronald E. Pelrine and ..."), and the family name sits within its first few words
40
+ # ("J. D. van der Waals").
41
+ _FIRST_AUTHOR_END = re.compile(r",|;|\(|\s(?:and|&)\s|\set\.?\s*al\b", re.IGNORECASE)
42
+ _FIRST_AUTHOR_TOKEN_WINDOW = 6
43
+ # Words that are neither title nor author in a citation such as
44
+ # "J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108, 1175 (1957)".
45
+ _CITATION_FILLER = {
46
+ "al", "and", "et", "vol", "no", "pp", "doi", "https", "http", "org", "dx", "art", "article",
47
+ "jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "sept", "oct", "nov", "dec",
48
+ }
49
+ # More words than this before the volume number are a title plus journal, not a journal alone.
50
+ _MAX_TITLELESS_JOURNAL_WORDS = 5
51
+ _JOURNAL_STOPWORDS = {"a", "an", "and", "de", "der", "des", "for", "in", "of", "on", "the", "und"}
38
52
  _DEFAULT_RATE_LIMIT_PAUSE_SECONDS = 15.0
39
53
  _MAX_RATE_LIMIT_PAUSE_SECONDS = 60.0
40
54
  _sleep = time.sleep
@@ -45,7 +59,14 @@ class ReferenceClient(Protocol):
45
59
  def fetch_work(self, doi: str) -> PaperRecord:
46
60
  ...
47
61
 
48
- def search_bibliographic(self, query: str, rows: int = 3) -> list[PaperRecord]:
62
+ def search_bibliographic(
63
+ self,
64
+ query: str,
65
+ rows: int = 3,
66
+ *,
67
+ author: str | None = None,
68
+ year_range: tuple[int, int] | None = None,
69
+ ) -> list[PaperRecord]:
49
70
  ...
50
71
 
51
72
  def registration_agency(self, doi: str) -> str | None:
@@ -147,7 +168,25 @@ def _check_doi_reference(entry: ReferenceEntry, doi: str, client: ReferenceClien
147
168
  if metadata.verdict == "PASS":
148
169
  status, reason = "VERIFIED", metadata.reason
149
170
  elif metadata.verdict == "WARN" and "metadata" in metadata.mismatches:
150
- if not entry.title and not text_title and _text_names_another_paper(entry.raw, fetched):
171
+ titleless = None if text_title else _titleless_differences(entry, fetched)
172
+ if titleless == []:
173
+ return ReferenceResult(
174
+ entry=entry,
175
+ status="VERIFIED",
176
+ verdict="PASS",
177
+ reason=(
178
+ "The reference has no article title; it matches CrossRef's record for this DOI on "
179
+ f"{_titleless_fields(fetched)}."
180
+ ),
181
+ fetched=fetched,
182
+ )
183
+ if titleless:
184
+ status = "MISMATCH"
185
+ reason = (
186
+ "The reference has no article title, and CrossRef's record for this DOI differs in "
187
+ f"{'; '.join(titleless)}."
188
+ )
189
+ elif not entry.title and not text_title and _text_names_another_paper(entry.raw, fetched):
151
190
  status = "MISMATCH"
152
191
  reason = (
153
192
  f"This DOI belongs to {_describe(fetched)}, which this reference does not mention; "
@@ -176,21 +215,13 @@ def _check_doi_reference(entry: ReferenceEntry, doi: str, client: ReferenceClien
176
215
 
177
216
 
178
217
  def _missing_from_crossref(entry: ReferenceEntry, doi: str, client: ReferenceClient) -> ReferenceResult:
179
- try:
180
- agency = client.registration_agency(doi)
181
- except Exception:
182
- agency = None
183
- # arXiv, Zenodo, and many Korean (KISTI) or Japanese (JaLC) DOIs are registered with
184
- # another agency, so a CrossRef 404 alone does not make them dead.
185
- if agency and agency.casefold() != "crossref":
218
+ agency = other_registration_agency(client, doi)
219
+ if agency:
186
220
  return ReferenceResult(
187
221
  entry=entry,
188
222
  status="UNVERIFIED",
189
223
  verdict="WARN",
190
- reason=(
191
- f"This DOI is registered with {agency}, not CrossRef, so its title and authors "
192
- f"were not compared; open https://doi.org/{doi} to confirm it is this work."
193
- ),
224
+ reason=not_in_crossref_reason(agency, doi),
194
225
  error_code="DOI_NOT_IN_CROSSREF",
195
226
  )
196
227
  return ReferenceResult(
@@ -216,7 +247,7 @@ def _insufficient_reason(provided: CitationInput, fetched: PaperRecord) -> str:
216
247
  if provided.title and not provided.first_author and fetched.authors:
217
248
  return (
218
249
  f"The DOI and title match CrossRef, but its first author ({fetched.authors[0]}) was not "
219
- "found in the reference; check the author names manually."
250
+ "found in the reference's first-author position; check the author names manually."
220
251
  )
221
252
  if provided.first_author and not provided.title:
222
253
  return (
@@ -258,7 +289,11 @@ def _resolve_reference(entry: ReferenceEntry, client: ReferenceClient) -> Refere
258
289
  # top three results ahead of the paper itself.
259
290
  candidates = client.search_bibliographic(query, rows=5) if query else []
260
291
  for candidate in candidates:
261
- if candidate.is_about_other_work or not _candidate_matches(entry, candidate):
292
+ if candidate.is_about_other_work:
293
+ continue
294
+ if not _candidate_matches(entry, candidate):
295
+ if _titleless_differences(entry, candidate) == []:
296
+ return _titleless_resolved(entry, candidate)
262
297
  continue
263
298
  if candidate.retraction_doi:
264
299
  return _retracted(entry, candidate, resolved_doi=candidate.doi)
@@ -282,6 +317,16 @@ def _resolve_reference(entry: ReferenceEntry, client: ReferenceClient) -> Refere
282
317
  resolved_doi=candidate.doi,
283
318
  fetched=candidate,
284
319
  )
320
+ structured = _structured_query(entry)
321
+ if structured:
322
+ # A short title-less citation ("A. G. Riess et al., Astron. J. 116, 1009 (1998).") can
323
+ # rank below unrelated records with similar numbers; a second search by first author,
324
+ # journal, volume, and page within the cited year finds it. Only full field agreement
325
+ # is accepted from it.
326
+ query, author, year_range = structured
327
+ for candidate in client.search_bibliographic(query, rows=5, author=author, year_range=year_range):
328
+ if not candidate.is_about_other_work and _titleless_differences(entry, candidate) == []:
329
+ return _titleless_resolved(entry, candidate)
285
330
  return ReferenceResult(
286
331
  entry=entry,
287
332
  status="UNVERIFIED",
@@ -291,6 +336,66 @@ def _resolve_reference(entry: ReferenceEntry, client: ReferenceClient) -> Refere
291
336
  )
292
337
 
293
338
 
339
+ def _titleless_resolved(entry: ReferenceEntry, candidate: PaperRecord) -> ReferenceResult:
340
+ if candidate.retraction_doi:
341
+ return _retracted(entry, candidate, resolved_doi=candidate.doi)
342
+ return ReferenceResult(
343
+ entry=entry,
344
+ status="RESOLVED",
345
+ verdict="PASS",
346
+ reason=(
347
+ f"Matched CrossRef record {candidate.doi} by bibliographic search; the reference has "
348
+ f"no article title, so it was matched on {_titleless_fields(candidate)}."
349
+ ),
350
+ error_code="REFERENCE_RESOLVED",
351
+ resolved_doi=candidate.doi,
352
+ fetched=candidate,
353
+ )
354
+
355
+
356
+ def _structured_query(entry: ReferenceEntry) -> tuple[str, str, tuple[int, int]] | None:
357
+ # The first author goes to `query.author`, the rest of the citation without the year (a
358
+ # bare year matches far too many records) to `query.bibliographic`, and the year (or the
359
+ # year before, for online-first papers) to a publication-date filter.
360
+ if entry.title or entry.year is None or not _looks_titleless(entry.raw):
361
+ return None
362
+ parts = _FIRST_AUTHOR_END.split(DOI_PATTERN.sub(" ", entry.raw), maxsplit=1)
363
+ names = re.findall(r"[^\W\d_]{2,}", parts[0])
364
+ if len(parts) < 2 or not names:
365
+ return None
366
+ rest = re.sub(r"\bet\.?\s*al\b\.?|\(?\b(?:1[89]|20)\d{2}[a-z]?\b\)?", " ", parts[1])
367
+ rest = " ".join(rest.split()).strip(" .,;")
368
+ if not rest:
369
+ return None
370
+ return rest, names[-1], (entry.year - 1, entry.year)
371
+
372
+
373
+ def _looks_titleless(text: str) -> bool:
374
+ # Without a record to compare against: the words before the volume number, once the
375
+ # authors are set aside, are only a journal name ("A. G. Riess et al., Astron. J. 116,
376
+ # 1009"), not a title plus journal. Authors are the first word and any word next to an
377
+ # initial ("Riess" in "A. G. Riess", "Cooper" in "L. N. Cooper", "Tang" in "Tang, C. W.").
378
+ text = re.sub(r"\bet\.?\s*al\b|\(?\b(?:1[89]|20)\d{2}[a-z]?\b\)?", " ", DOI_PATTERN.sub(" ", text))
379
+ tokens = re.findall(r"[^\W_]+", text)
380
+ numbers = [index for index, token in enumerate(tokens) if token.isdigit()]
381
+ # Volume and page (or article number) must both follow the journal.
382
+ if len(numbers) < 2:
383
+ return False
384
+ before = tokens[: numbers[0]]
385
+ initials = {index for index, token in enumerate(before) if token.isalpha() and token.isupper() and len(token) <= 2}
386
+ words = [
387
+ token
388
+ for index, token in enumerate(before)
389
+ if index not in initials
390
+ and index != 0
391
+ and not ({index - 1, index + 1} & initials)
392
+ and len(token) >= 3
393
+ and token.isalpha()
394
+ and token.casefold() not in _JOURNAL_STOPWORDS | _CITATION_FILLER
395
+ ]
396
+ return len(words) <= _MAX_TITLELESS_JOURNAL_WORDS
397
+
398
+
294
399
  def _bibliographic_query(entry: ReferenceEntry) -> str:
295
400
  if entry.title:
296
401
  parts = (entry.title, entry.first_author, entry.year, entry.journal)
@@ -326,11 +431,98 @@ def _first_author_matches(entry: ReferenceEntry, candidate: PaperRecord) -> bool
326
431
  def _first_author_in_text(text: str, record: PaperRecord) -> str | None:
327
432
  if not text or not record.authors:
328
433
  return None
329
- if hangul_surname_matches(text, record.authors[0]):
330
- return record.authors[0]
331
- family = author_tokens(record.authors[0])
332
- if family and family[-1] in author_tokens(text)[:_FIRST_AUTHOR_TOKEN_WINDOW]:
333
- return record.authors[0]
434
+ first_author = record.authors[0]
435
+ if hangul_surname_matches(text, first_author):
436
+ return first_author
437
+ family = author_tokens(first_author)
438
+ if not family:
439
+ return None
440
+ if looks_like_group_author(family):
441
+ # A group author ("Writing Group for the ... Investigators") opens the reference whole.
442
+ group = [token for token in family if token != "the"]
443
+ opening = [token for token in author_tokens(text) if token != "the"][: len(group)]
444
+ return first_author if opening == group else None
445
+ # Only the first author's own name counts, so "Perlmutter S, Riess AG" does not pass for
446
+ # a paper whose first author is Riess.
447
+ first_name = _FIRST_AUTHOR_END.split(text, maxsplit=1)[0]
448
+ if family[-1] in author_tokens(first_name)[:_FIRST_AUTHOR_TOKEN_WINDOW]:
449
+ return first_author
450
+ return None
451
+
452
+
453
+ def _titleless_differences(entry: ReferenceEntry, record: PaperRecord) -> list[str] | None:
454
+ # A citation without an article title ("Phys. Rev. 108, 1175 (1957)") is compared on the
455
+ # fields it does carry. None means it carries a title (words that are not authors,
456
+ # journal, or numbers) or the record has no volume or page to compare; otherwise the list
457
+ # names the fields that disagree, and an empty list means every field agrees.
458
+ if entry.title or not (record.volume or record.first_page):
459
+ return None
460
+ tokens = _plain_tokens(DOI_PATTERN.sub(" ", entry.raw))
461
+ journal_run = next(
462
+ (
463
+ run
464
+ for journal in (record.journal, *record.journal_abbreviations)
465
+ if journal and (run := _journal_run(journal, tokens))
466
+ ),
467
+ None,
468
+ )
469
+ author_words = {token for author in record.authors for token in author_tokens(author)}
470
+ leftover = [
471
+ token
472
+ for index, token in enumerate(tokens)
473
+ if not (journal_run and index in journal_run)
474
+ and token not in author_words
475
+ and token not in _CITATION_FILLER
476
+ and not token.isdigit()
477
+ and len(token) >= 3
478
+ ]
479
+ if len(leftover) >= 3:
480
+ return None
481
+ differences = []
482
+ if not journal_run:
483
+ differences.append(f"journal (CrossRef: {record.journal or 'none listed'})")
484
+ if record.volume and record.volume.casefold() not in tokens:
485
+ differences.append(f"volume (CrossRef: {record.volume})")
486
+ if record.first_page and record.first_page.casefold() not in tokens:
487
+ differences.append(f"first page (CrossRef: {record.first_page})")
488
+ years = record_years(record)
489
+ if entry.year is None or entry.year not in years:
490
+ reference = f"reference: {entry.year}; " if entry.year is not None else ""
491
+ crossref_years = "/".join(str(year) for year in years) or "none listed"
492
+ differences.append(f"year ({reference}CrossRef: {crossref_years})")
493
+ if _first_author_in_text(entry.raw, record) is None:
494
+ differences.append(f"first author (CrossRef: {record.authors[0] if record.authors else 'none listed'})")
495
+ return differences
496
+
497
+
498
+ def _titleless_fields(record: PaperRecord) -> str:
499
+ fields = ["journal"]
500
+ if record.volume:
501
+ fields.append("volume")
502
+ if record.first_page:
503
+ fields.append("first page")
504
+ fields.extend(["year", "first author"])
505
+ return ", ".join(fields[:-1]) + f", and {fields[-1]}"
506
+
507
+
508
+ def _plain_tokens(text: str) -> list[str]:
509
+ folded = "".join(
510
+ char for char in unicodedata.normalize("NFKD", text.casefold()) if not unicodedata.combining(char)
511
+ )
512
+ return re.findall(r"[a-z0-9]+", folded)
513
+
514
+
515
+ def _journal_run(journal: str, tokens: list[str]) -> set[int] | None:
516
+ # Full name or a standard abbreviation: "Phys. Rev. Lett." matches "Physical Review
517
+ # Letters" word by word, each abbreviated word being the start of the full one.
518
+ words = [word for word in _plain_tokens(journal) if word not in _JOURNAL_STOPWORDS]
519
+ if not words:
520
+ return None
521
+ positions = [index for index, token in enumerate(tokens) if token not in _JOURNAL_STOPWORDS]
522
+ for start in range(len(positions) - len(words) + 1):
523
+ window = positions[start : start + len(words)]
524
+ if all(words[offset].startswith(tokens[index]) for offset, index in enumerate(window)):
525
+ return set(range(window[0], window[-1] + 1))
334
526
  return None
335
527
 
336
528