ref-verify 1.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ref_verify/cli.py ADDED
@@ -0,0 +1,162 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from typing import Sequence
7
+
8
+ from ref_verify.abstract_lookup import (
9
+ AbstractSourceClient,
10
+ lookup_abstract,
11
+ lookup_selected_abstract,
12
+ )
13
+ from ref_verify.claim_check import check_claim_support
14
+ from ref_verify.crossref import CrossrefClient
15
+ from ref_verify.doi_check import normalize_doi, verify_doi_metadata
16
+ from ref_verify.models import CitationInput, ClaimSupportResult
17
+ from ref_verify.pubmed import PubMedClient
18
+ from ref_verify.semantic_scholar import SemanticScholarClient
19
+
20
+
21
+ def main(
22
+ argv: Sequence[str] | None = None,
23
+ *,
24
+ client: CrossrefClient | None = None,
25
+ abstract_clients: Sequence[AbstractSourceClient] | None = None,
26
+ ) -> int:
27
+ parser = _build_parser()
28
+ args = parser.parse_args(argv)
29
+ lookup_client = client or CrossrefClient()
30
+ fallback_clients = list(abstract_clients) if abstract_clients is not None else _default_abstract_clients()
31
+
32
+ try:
33
+ if args.command == "verify-doi":
34
+ return _verify_doi(args, lookup_client)
35
+ if args.command == "check-claim":
36
+ return _check_claim(args, lookup_client, fallback_clients)
37
+ except Exception as exc:
38
+ _emit({"error": str(exc)}, as_json=getattr(args, "json", False))
39
+ return 1
40
+
41
+ parser.print_help()
42
+ return 2
43
+
44
+
45
+ def _build_parser() -> argparse.ArgumentParser:
46
+ parser = argparse.ArgumentParser(
47
+ prog="ref-verify",
48
+ description="Verify citation metadata and abstract-grounded claims.",
49
+ )
50
+ subparsers = parser.add_subparsers(dest="command")
51
+
52
+ verify = subparsers.add_parser("verify-doi", help="Check DOI metadata")
53
+ verify.add_argument("doi")
54
+ verify.add_argument("--title")
55
+ verify.add_argument("--first-author")
56
+ verify.add_argument("--year", type=int)
57
+ verify.add_argument("--json", action="store_true")
58
+
59
+ claim = subparsers.add_parser("check-claim", help="Check a claim against a DOI abstract")
60
+ claim.add_argument("doi")
61
+ claim.add_argument("--claim", required=True)
62
+ claim.add_argument(
63
+ "--source",
64
+ choices=("auto", "crossref", "semantic-scholar", "pubmed"),
65
+ default="auto",
66
+ help="Select an abstract source for debugging; default tries DOI-bound fallback sources.",
67
+ )
68
+ claim.add_argument("--json", action="store_true")
69
+
70
+ return parser
71
+
72
+
73
+ def _verify_doi(args: argparse.Namespace, client: CrossrefClient) -> int:
74
+ lookup_doi = normalize_doi(args.doi)
75
+ fetched = client.fetch_work(lookup_doi)
76
+ provided = CitationInput(
77
+ doi=args.doi,
78
+ title=args.title,
79
+ first_author=args.first_author,
80
+ year=args.year,
81
+ )
82
+ result = verify_doi_metadata(provided, fetched)
83
+ _emit(result.to_dict(), as_json=args.json)
84
+ return 0 if result.verdict == "PASS" else 2
85
+
86
+
87
+ def _check_claim(
88
+ args: argparse.Namespace,
89
+ client: CrossrefClient,
90
+ fallback_clients: Sequence[AbstractSourceClient],
91
+ ) -> int:
92
+ lookup_doi = normalize_doi(args.doi)
93
+ selected_clients = _select_abstract_clients(fallback_clients, args.source)
94
+ if args.source in ("auto", "crossref"):
95
+ fetched = client.fetch_work(lookup_doi)
96
+ lookup_result = lookup_abstract(lookup_doi, fetched, selected_clients)
97
+ else:
98
+ lookup_result = lookup_selected_abstract(lookup_doi, selected_clients)
99
+ if lookup_result.error_code == "DOI_MISMATCH":
100
+ result = ClaimSupportResult(
101
+ status="UNVERIFIABLE",
102
+ verdict="WARN",
103
+ reason="Fetched DOI does not match the requested DOI.",
104
+ evidence="",
105
+ paper=lookup_result.record,
106
+ claim=args.claim,
107
+ )
108
+ _emit(_claim_payload(result, lookup_result), as_json=args.json)
109
+ return 2
110
+
111
+ result = check_claim_support(lookup_result.record, args.claim)
112
+ _emit(_claim_payload(result, lookup_result), as_json=args.json)
113
+ return 0 if result.verdict == "ACCEPT" else 2
114
+
115
+
116
+ def _claim_payload(result: ClaimSupportResult, lookup_result) -> dict:
117
+ payload = result.to_dict()
118
+ payload["abstract_source"] = lookup_result.abstract_source
119
+ payload["source_attempts"] = [attempt.to_dict() for attempt in lookup_result.attempts]
120
+ payload["error_code"] = lookup_result.error_code or _claim_error_code(result)
121
+ return payload
122
+
123
+
124
+ def _claim_error_code(result: ClaimSupportResult) -> str:
125
+ if result.verdict == "ACCEPT":
126
+ return "CLAIM_SUPPORTED"
127
+ if result.status == "UNVERIFIABLE":
128
+ return "NO_ABSTRACT"
129
+ if result.status == "PARTIAL":
130
+ if "does not explicitly support" in result.reason:
131
+ return "CLAIM_NOT_EXPLICIT"
132
+ return "CLAIM_AMBIGUOUS"
133
+ return "CLAIM_NOT_EXPLICIT"
134
+
135
+
136
+ def _default_abstract_clients() -> list[AbstractSourceClient]:
137
+ return [SemanticScholarClient(), PubMedClient()]
138
+
139
+
140
+ def _select_abstract_clients(
141
+ fallback_clients: Sequence[AbstractSourceClient],
142
+ source: str,
143
+ ) -> Sequence[AbstractSourceClient]:
144
+ if source in ("auto", "crossref"):
145
+ return [] if source == "crossref" else fallback_clients
146
+ source_name = source.replace("-", "_")
147
+ return [client for client in fallback_clients if client.source_name == source_name]
148
+
149
+
150
+ def _emit(payload: dict, *, as_json: bool) -> None:
151
+ if as_json:
152
+ print(json.dumps(payload, indent=2, sort_keys=True))
153
+ return
154
+ if "error" in payload:
155
+ print(f"ERROR: {payload['error']}")
156
+ return
157
+ for key, value in payload.items():
158
+ print(f"{key}: {value}")
159
+
160
+
161
+ if __name__ == "__main__":
162
+ sys.exit(main())
ref_verify/crossref.py ADDED
@@ -0,0 +1,89 @@
1
+ from __future__ import annotations
2
+
3
+ import html
4
+ import json
5
+ import re
6
+ from typing import Any
7
+ from urllib.parse import quote
8
+ from urllib.request import Request, urlopen
9
+
10
+ from ref_verify import __version__
11
+ from ref_verify.doi_check import normalize_doi
12
+ from ref_verify.models import PaperRecord
13
+
14
+
15
+ class CrossrefClient:
16
+ def __init__(self, timeout: float = 20.0) -> None:
17
+ self.timeout = timeout
18
+
19
+ def fetch_work(self, doi: str) -> PaperRecord:
20
+ encoded_doi = quote(normalize_doi(doi), safe="")
21
+ request = Request(
22
+ f"https://api.crossref.org/works/{encoded_doi}",
23
+ headers={
24
+ "User-Agent": (
25
+ f"ref-verify/{__version__} "
26
+ "(+https://github.com/Moonweave-Research/ref-verify)"
27
+ )
28
+ },
29
+ )
30
+ with urlopen(request, timeout=self.timeout) as response:
31
+ payload = json.loads(response.read().decode("utf-8"))
32
+ return parse_crossref_work(payload["message"])
33
+
34
+
35
+ def parse_crossref_work(message: dict[str, Any]) -> PaperRecord:
36
+ doi = str(message.get("DOI") or "")
37
+ title = _first_string(message.get("title")) or "[title missing]"
38
+ authors = [
39
+ author_name
40
+ for author in message.get("author", [])
41
+ if (author_name := _crossref_author_name(author))
42
+ ]
43
+ year = _published_year(message)
44
+ journal = _first_string(message.get("container-title"))
45
+ abstract = _clean_abstract(message.get("abstract"))
46
+ url = message.get("URL")
47
+
48
+ return PaperRecord(
49
+ doi=doi,
50
+ title=title,
51
+ authors=authors,
52
+ year=year,
53
+ abstract=abstract,
54
+ source="CrossRef",
55
+ journal=journal,
56
+ url=str(url) if url else None,
57
+ )
58
+
59
+
60
+ def _first_string(value: Any) -> str | None:
61
+ if isinstance(value, list) and value:
62
+ return str(value[0]).strip()
63
+ if isinstance(value, str) and value.strip():
64
+ return value.strip()
65
+ return None
66
+
67
+
68
+ def _crossref_author_name(author: Any) -> str:
69
+ if not isinstance(author, dict):
70
+ return ""
71
+ family = str(author.get("family") or "").strip()
72
+ if family:
73
+ return family
74
+ return str(author.get("name") or "").strip()
75
+
76
+
77
+ def _published_year(message: dict[str, Any]) -> int | None:
78
+ for key in ("published-print", "published-online", "published", "issued"):
79
+ date_parts = message.get(key, {}).get("date-parts")
80
+ if date_parts and date_parts[0]:
81
+ return int(date_parts[0][0])
82
+ return None
83
+
84
+
85
+ def _clean_abstract(value: Any) -> str | None:
86
+ if not isinstance(value, str) or not value.strip():
87
+ return None
88
+ without_tags = re.sub(r"<[^>]+>", " ", value)
89
+ return re.sub(r"\s+", " ", html.unescape(without_tags)).strip()
@@ -0,0 +1,191 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ import unicodedata
5
+ from urllib.parse import unquote
6
+
7
+ from ref_verify.models import CitationInput, MetadataCheckResult, PaperRecord
8
+
9
+ _GREEK_LETTER_NAMES = {
10
+ "\u03b1": "alpha",
11
+ "\u03b2": "beta",
12
+ "\u03b3": "gamma",
13
+ "\u03b4": "delta",
14
+ "\u03b5": "epsilon",
15
+ "\u03b6": "zeta",
16
+ "\u03b7": "eta",
17
+ "\u03b8": "theta",
18
+ "\u03b9": "iota",
19
+ "\u03ba": "kappa",
20
+ "\u03bb": "lambda",
21
+ "\u03bc": "mu",
22
+ "\u03bd": "nu",
23
+ "\u03be": "xi",
24
+ "\u03bf": "omicron",
25
+ "\u03c0": "pi",
26
+ "\u03c1": "rho",
27
+ "\u03c2": "sigma",
28
+ "\u03c3": "sigma",
29
+ "\u03c4": "tau",
30
+ "\u03c5": "upsilon",
31
+ "\u03c6": "phi",
32
+ "\u03c7": "chi",
33
+ "\u03c8": "psi",
34
+ "\u03c9": "omega",
35
+ }
36
+
37
+ _GROUP_AUTHOR_TERMS = {
38
+ "association",
39
+ "collaboration",
40
+ "committee",
41
+ "consortium",
42
+ "group",
43
+ "network",
44
+ "organisation",
45
+ "organization",
46
+ "society",
47
+ "team",
48
+ "working",
49
+ }
50
+
51
+
52
+ def verify_doi_metadata(
53
+ provided: CitationInput,
54
+ fetched: PaperRecord,
55
+ ) -> MetadataCheckResult:
56
+ mismatches: list[str] = []
57
+
58
+ if provided.doi and not doi_matches(provided.doi, fetched.doi):
59
+ mismatches.append("doi")
60
+
61
+ if not _has_comparison_metadata(provided):
62
+ mismatches.append("metadata")
63
+
64
+ if provided.title and not _titles_match(provided.title, fetched.title):
65
+ mismatches.append("title")
66
+
67
+ if provided.first_author and not _author_matches(
68
+ provided.first_author,
69
+ fetched.authors[0] if fetched.authors else None,
70
+ ):
71
+ mismatches.append("first_author")
72
+
73
+ if provided.year is not None and (
74
+ fetched.year is None or provided.year != fetched.year
75
+ ):
76
+ mismatches.append("year")
77
+
78
+ if not mismatches:
79
+ verdict = "PASS"
80
+ reason = "Provided citation metadata matches the fetched CrossRef record."
81
+ elif any(field in mismatches for field in ("doi", "title", "first_author")):
82
+ verdict = "REJECT"
83
+ reason = "DOI resolves to a materially different paper than provided."
84
+ elif "metadata" in mismatches:
85
+ verdict = "WARN"
86
+ reason = "Insufficient citation metadata was provided to verify the fetched CrossRef record."
87
+ else:
88
+ verdict = "WARN"
89
+ reason = "DOI resolves, but minor metadata differs from the provided citation."
90
+
91
+ return MetadataCheckResult(
92
+ verdict=verdict,
93
+ mismatches=mismatches,
94
+ reason=reason,
95
+ provided=provided,
96
+ fetched=fetched,
97
+ )
98
+
99
+
100
+ def _has_comparison_metadata(provided: CitationInput) -> bool:
101
+ return bool(provided.title and provided.first_author)
102
+
103
+
104
+ def doi_matches(provided: str, fetched: str) -> bool:
105
+ return normalize_doi(provided) == normalize_doi(fetched)
106
+
107
+
108
+ def normalize_doi(value: str) -> str:
109
+ normalized = value.strip().casefold()
110
+ normalized = re.sub(r"^(?:https?://)?(?:dx\.)?doi\.org/", "", normalized)
111
+ normalized = re.sub(r"^doi:\s*", "", normalized)
112
+ normalized = unquote(normalized)
113
+ return _strip_trailing_doi_punctuation(normalized)
114
+
115
+
116
+ def _strip_trailing_doi_punctuation(value: str) -> str:
117
+ stripped = value.strip()
118
+ while stripped:
119
+ without_sentence_punctuation = stripped.rstrip(".,;:")
120
+ if without_sentence_punctuation != stripped:
121
+ stripped = without_sentence_punctuation.rstrip()
122
+ continue
123
+ if stripped.endswith(")") and stripped.count(")") > stripped.count("("):
124
+ stripped = stripped[:-1].rstrip()
125
+ continue
126
+ return stripped
127
+ return stripped
128
+
129
+
130
+ def _titles_match(provided: str, fetched: str) -> bool:
131
+ if _numbers(provided) != _numbers(fetched):
132
+ return False
133
+
134
+ return _title_tokens(provided) == _title_tokens(fetched)
135
+
136
+
137
+ def _author_matches(provided: str, fetched: str | None) -> bool:
138
+ if not fetched:
139
+ return False
140
+ provided_tokens = _author_tokens(provided)
141
+ fetched_tokens = _author_tokens(fetched)
142
+ if not provided_tokens or not fetched_tokens:
143
+ return False
144
+ if _looks_like_group_author(provided_tokens) or _looks_like_group_author(
145
+ fetched_tokens,
146
+ ):
147
+ return provided_tokens == fetched_tokens
148
+ return provided_tokens[-1] == fetched_tokens[-1]
149
+
150
+
151
+ def _author_tokens(value: str) -> list[str]:
152
+ normalized = _strip_diacritics(value.casefold())
153
+ cleaned = re.sub(r"[^a-z -]", " ", normalized).strip()
154
+ return [part for part in re.split(r"\s+", cleaned) if part]
155
+
156
+
157
+ def _looks_like_group_author(tokens: list[str]) -> bool:
158
+ return bool(set(tokens) & _GROUP_AUTHOR_TERMS)
159
+
160
+
161
+ def _numbers(value: str) -> list[str]:
162
+ return [
163
+ number.replace(",", "")
164
+ for number in re.findall(r"\d+(?:,\d{3})*(?:\.\d+)?", value)
165
+ ]
166
+
167
+
168
+ def _title_tokens(value: str) -> list[str]:
169
+ normalized = _transliterate_greek_letters(_strip_diacritics(value.casefold()))
170
+ return [_singularize(token) for token in re.findall(r"[^\W_]+", normalized)]
171
+
172
+
173
+ def _singularize(token: str) -> str:
174
+ if token.endswith("s") and len(token) > 3:
175
+ return token[:-1]
176
+ return token
177
+
178
+
179
+ def _strip_diacritics(value: str) -> str:
180
+ return "".join(
181
+ char
182
+ for char in unicodedata.normalize("NFKD", value)
183
+ if not unicodedata.combining(char)
184
+ )
185
+
186
+
187
+ def _transliterate_greek_letters(value: str) -> str:
188
+ return "".join(
189
+ f" {_GREEK_LETTER_NAMES[char]} " if char in _GREEK_LETTER_NAMES else char
190
+ for char in value
191
+ )
ref_verify/models.py ADDED
@@ -0,0 +1,89 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import asdict, dataclass
4
+ from typing import Any
5
+
6
+
7
+ @dataclass(frozen=True)
8
+ class CitationInput:
9
+ doi: str
10
+ title: str | None = None
11
+ first_author: str | None = None
12
+ year: int | None = None
13
+
14
+ def to_dict(self) -> dict[str, Any]:
15
+ return asdict(self)
16
+
17
+
18
+ @dataclass(frozen=True)
19
+ class PaperRecord:
20
+ doi: str
21
+ title: str
22
+ authors: list[str]
23
+ year: int | None
24
+ abstract: str | None
25
+ source: str
26
+ journal: str | None = None
27
+ url: str | None = None
28
+
29
+ def to_dict(self) -> dict[str, Any]:
30
+ return asdict(self)
31
+
32
+
33
+ @dataclass(frozen=True)
34
+ class MetadataCheckResult:
35
+ verdict: str
36
+ mismatches: list[str]
37
+ reason: str
38
+ provided: CitationInput
39
+ fetched: PaperRecord
40
+
41
+ def to_dict(self) -> dict[str, Any]:
42
+ payload = asdict(self)
43
+ payload["provided"] = self.provided.to_dict()
44
+ payload["fetched"] = self.fetched.to_dict()
45
+ return payload
46
+
47
+
48
+ @dataclass(frozen=True)
49
+ class ClaimSupportResult:
50
+ status: str
51
+ verdict: str
52
+ reason: str
53
+ evidence: str
54
+ paper: PaperRecord
55
+ claim: str
56
+
57
+ def to_dict(self) -> dict[str, Any]:
58
+ payload = asdict(self)
59
+ payload["paper"] = self.paper.to_dict()
60
+ return payload
61
+
62
+
63
+ @dataclass(frozen=True)
64
+ class AbstractSourceAttempt:
65
+ source: str
66
+ status: str
67
+ reason: str
68
+ record_id: str | None = None
69
+ doi: str | None = None
70
+ elapsed_ms: int | None = None
71
+
72
+ def to_dict(self) -> dict[str, Any]:
73
+ return asdict(self)
74
+
75
+
76
+ @dataclass(frozen=True)
77
+ class AbstractLookupResult:
78
+ record: PaperRecord
79
+ abstract_source: str | None
80
+ attempts: list[AbstractSourceAttempt]
81
+ error_code: str | None = None
82
+
83
+ def to_dict(self) -> dict[str, Any]:
84
+ return {
85
+ "record": self.record.to_dict(),
86
+ "abstract_source": self.abstract_source,
87
+ "source_attempts": [attempt.to_dict() for attempt in self.attempts],
88
+ "error_code": self.error_code,
89
+ }