ref-verify 1.2.0__tar.gz → 1.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ref_verify-1.2.0/src/ref_verify.egg-info → ref_verify-1.2.2}/PKG-INFO +31 -14
- {ref_verify-1.2.0 → ref_verify-1.2.2}/README.md +30 -13
- {ref_verify-1.2.0 → ref_verify-1.2.2}/pyproject.toml +1 -1
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/__init__.py +1 -1
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/claim_check.py +16 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/cli.py +55 -3
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/crossref.py +15 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/doi_check.py +17 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/models.py +1 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/numeric_claim.py +49 -3
- {ref_verify-1.2.0 → ref_verify-1.2.2/src/ref_verify.egg-info}/PKG-INFO +31 -14
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_cli.py +72 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_crossref.py +27 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_doi_check.py +65 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_numeric_claim.py +40 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_skill_docs.py +52 -2
- {ref_verify-1.2.0 → ref_verify-1.2.2}/LICENSE +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/setup.cfg +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/abstract_lookup.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/batch.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/openalex.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/pubmed.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify/semantic_scholar.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify.egg-info/SOURCES.txt +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify.egg-info/dependency_links.txt +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify.egg-info/entry_points.txt +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/src/ref_verify.egg-info/top_level.txt +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_abstract_sources.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_batch.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_claim_check.py +0 -0
- {ref_verify-1.2.0 → ref_verify-1.2.2}/tests/test_package_smoke.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ref-verify
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: Executable DOI and claim verification helpers for academic citations
|
|
5
5
|
Author: Moonweave Research
|
|
6
6
|
License-Expression: MIT
|
|
@@ -45,6 +45,8 @@ Works with **Claude Code, Cursor, Codex**, and any agent that supports the
|
|
|
45
45
|
|
|
46
46
|
After installation, use it like a normal agent skill. You do not start a server and you do not configure MCP for this workflow. No MCP server is required for this workflow.
|
|
47
47
|
|
|
48
|
+
The skill includes its own copy of the CLI engine, and the agent runs it from the skill folder, so nothing else needs to be installed. Python 3.10 or newer must be available as `python3`.
|
|
49
|
+
|
|
48
50
|
For explicit agent tool-calling rules, see [AGENT_USAGE.md](https://github.com/Moonweave-Research/ref-verify/blob/main/AGENT_USAGE.md).
|
|
49
51
|
|
|
50
52
|
---
|
|
@@ -85,13 +87,20 @@ checks that are currently safe to automate directly:
|
|
|
85
87
|
- JSON output for agent-readable routing
|
|
86
88
|
- Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
|
|
87
89
|
|
|
88
|
-
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol.
|
|
90
|
+
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol. The CLI rejects a DOI that CrossRef records as retracted (via its retraction notice); retraction banners CrossRef does not know about, Unpaywall, arXiv, and two-source existence checks remain in the `SKILL.md` protocol.
|
|
89
91
|
|
|
90
92
|
The CLI has zero third-party Python runtime dependencies, but it is not an
|
|
91
93
|
offline verifier. Functional checks require outbound HTTPS access to public
|
|
92
94
|
academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
|
|
93
95
|
|
|
94
|
-
|
|
96
|
+
To run the CLI yourself, install it from PyPI:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
uvx ref-verify --help # run without installing (uv)
|
|
100
|
+
pipx install ref-verify # or install the `ref-verify` command
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Or install it from a local checkout:
|
|
95
104
|
|
|
96
105
|
```bash
|
|
97
106
|
git clone https://github.com/Moonweave-Research/ref-verify.git
|
|
@@ -172,20 +181,20 @@ normal CI does not fail because an upstream API is temporarily unavailable.
|
|
|
172
181
|
|
|
173
182
|
---
|
|
174
183
|
|
|
175
|
-
## Scope —
|
|
184
|
+
## Scope — optional CLI versus manual audit
|
|
176
185
|
|
|
177
186
|
`ref-verify` is a conservative guard, not an oracle. It errs toward flagging: an
|
|
178
187
|
`ACCEPT` is high-confidence, and **anything else means "not auto-verifiable —
|
|
179
188
|
check it yourself", not "the citation is wrong."**
|
|
180
189
|
|
|
181
|
-
**
|
|
190
|
+
**The optional CLI verifies**
|
|
182
191
|
|
|
183
192
|
- DOI metadata: title, first-author surname, and year against CrossRef.
|
|
184
193
|
- Whether a DOI-bound **abstract** explicitly supports a specific numeric or
|
|
185
194
|
literal claim, quoted verbatim. If no abstract is reachable, it returns
|
|
186
195
|
`UNVERIFIABLE` rather than guessing.
|
|
187
196
|
|
|
188
|
-
**
|
|
197
|
+
**The optional CLI does not verify** (out of scope by design, not bugs)
|
|
189
198
|
|
|
190
199
|
- **Full-text, figure, table, or supplementary values** — abstract-only. A number
|
|
191
200
|
that appears only in the body stays `UNVERIFIABLE`.
|
|
@@ -199,7 +208,13 @@ check it yourself", not "the citation is wrong."**
|
|
|
199
208
|
- **Paper quality, novelty, field consensus**, or whether the *full* paper
|
|
200
209
|
supports a broader statement.
|
|
201
210
|
|
|
202
|
-
|
|
211
|
+
The agent skill's manual Full Audit protocol goes beyond the optional CLI for
|
|
212
|
+
mechanism, implementation, and procedural claims. It requires a fetched
|
|
213
|
+
full-text passage at that source depth; when full text is unavailable, it
|
|
214
|
+
returns `WARN (ABSTRACT-ONLY)` instead of upgrading an abstract topic match to
|
|
215
|
+
`ACCEPT`.
|
|
216
|
+
|
|
217
|
+
**Reading a CLI verdict**
|
|
203
218
|
|
|
204
219
|
| Verdict | Meaning |
|
|
205
220
|
|---|---|
|
|
@@ -224,8 +239,9 @@ exit code, so weak or mismatched metadata cannot silently pass automation gates.
|
|
|
224
239
|
|
|
225
240
|
**Full Audit** is for literature search and final pre-submission review. The
|
|
226
241
|
skill fetches abstracts through CrossRef, OpenAlex, Semantic Scholar, Unpaywall,
|
|
227
|
-
arXiv, and PubMed where needed
|
|
228
|
-
claim
|
|
242
|
+
arXiv, and PubMed where needed. For a topline claim, it checks the abstract; for
|
|
243
|
+
a mechanism, implementation, or procedural claim, it continues to a fetched
|
|
244
|
+
full-text passage before assigning support.
|
|
229
245
|
|
|
230
246
|
For a single DOI-backed claim, the CLI can run the abstract check:
|
|
231
247
|
|
|
@@ -266,13 +282,15 @@ Current `check-claim` error codes:
|
|
|
266
282
|
- `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
|
|
267
283
|
- `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
|
|
268
284
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
269
|
-
- `DOI_NOT_FOUND`: selected source did not find a DOI-bound record.
|
|
285
|
+
- `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
|
|
286
|
+
- `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
|
|
270
287
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
271
288
|
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
272
289
|
|
|
273
290
|
> Core rule: every content statement about a paper must come from a live-fetched
|
|
274
|
-
>
|
|
275
|
-
>
|
|
291
|
+
> source at the depth the claim requires — abstract for topline claims, full
|
|
292
|
+
> text for mechanism, implementation, or procedural claims. If the required
|
|
293
|
+
> source is inaccessible, say so. Do not fill the gap from memory.
|
|
276
294
|
|
|
277
295
|
---
|
|
278
296
|
|
|
@@ -309,5 +327,4 @@ that as `WARN (PARTIAL)` instead of accepting the citation.
|
|
|
309
327
|
|
|
310
328
|
## Related
|
|
311
329
|
|
|
312
|
-
- [
|
|
313
|
-
- [decide-skill](https://github.com/Moonweave-Systems/decide-skill) - decision automation for non-expert domains
|
|
330
|
+
- [decision-kernel](https://github.com/Moonweave-Systems/decision-kernel) - evidence-gated decisions and drift/done checks for coding agents
|
|
@@ -34,6 +34,8 @@ Works with **Claude Code, Cursor, Codex**, and any agent that supports the
|
|
|
34
34
|
|
|
35
35
|
After installation, use it like a normal agent skill. You do not start a server and you do not configure MCP for this workflow. No MCP server is required for this workflow.
|
|
36
36
|
|
|
37
|
+
The skill includes its own copy of the CLI engine, and the agent runs it from the skill folder, so nothing else needs to be installed. Python 3.10 or newer must be available as `python3`.
|
|
38
|
+
|
|
37
39
|
For explicit agent tool-calling rules, see [AGENT_USAGE.md](https://github.com/Moonweave-Research/ref-verify/blob/main/AGENT_USAGE.md).
|
|
38
40
|
|
|
39
41
|
---
|
|
@@ -74,13 +76,20 @@ checks that are currently safe to automate directly:
|
|
|
74
76
|
- JSON output for agent-readable routing
|
|
75
77
|
- Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
|
|
76
78
|
|
|
77
|
-
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol.
|
|
79
|
+
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol. The CLI rejects a DOI that CrossRef records as retracted (via its retraction notice); retraction banners CrossRef does not know about, Unpaywall, arXiv, and two-source existence checks remain in the `SKILL.md` protocol.
|
|
78
80
|
|
|
79
81
|
The CLI has zero third-party Python runtime dependencies, but it is not an
|
|
80
82
|
offline verifier. Functional checks require outbound HTTPS access to public
|
|
81
83
|
academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
|
|
82
84
|
|
|
83
|
-
|
|
85
|
+
To run the CLI yourself, install it from PyPI:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
uvx ref-verify --help # run without installing (uv)
|
|
89
|
+
pipx install ref-verify # or install the `ref-verify` command
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Or install it from a local checkout:
|
|
84
93
|
|
|
85
94
|
```bash
|
|
86
95
|
git clone https://github.com/Moonweave-Research/ref-verify.git
|
|
@@ -161,20 +170,20 @@ normal CI does not fail because an upstream API is temporarily unavailable.
|
|
|
161
170
|
|
|
162
171
|
---
|
|
163
172
|
|
|
164
|
-
## Scope —
|
|
173
|
+
## Scope — optional CLI versus manual audit
|
|
165
174
|
|
|
166
175
|
`ref-verify` is a conservative guard, not an oracle. It errs toward flagging: an
|
|
167
176
|
`ACCEPT` is high-confidence, and **anything else means "not auto-verifiable —
|
|
168
177
|
check it yourself", not "the citation is wrong."**
|
|
169
178
|
|
|
170
|
-
**
|
|
179
|
+
**The optional CLI verifies**
|
|
171
180
|
|
|
172
181
|
- DOI metadata: title, first-author surname, and year against CrossRef.
|
|
173
182
|
- Whether a DOI-bound **abstract** explicitly supports a specific numeric or
|
|
174
183
|
literal claim, quoted verbatim. If no abstract is reachable, it returns
|
|
175
184
|
`UNVERIFIABLE` rather than guessing.
|
|
176
185
|
|
|
177
|
-
**
|
|
186
|
+
**The optional CLI does not verify** (out of scope by design, not bugs)
|
|
178
187
|
|
|
179
188
|
- **Full-text, figure, table, or supplementary values** — abstract-only. A number
|
|
180
189
|
that appears only in the body stays `UNVERIFIABLE`.
|
|
@@ -188,7 +197,13 @@ check it yourself", not "the citation is wrong."**
|
|
|
188
197
|
- **Paper quality, novelty, field consensus**, or whether the *full* paper
|
|
189
198
|
supports a broader statement.
|
|
190
199
|
|
|
191
|
-
|
|
200
|
+
The agent skill's manual Full Audit protocol goes beyond the optional CLI for
|
|
201
|
+
mechanism, implementation, and procedural claims. It requires a fetched
|
|
202
|
+
full-text passage at that source depth; when full text is unavailable, it
|
|
203
|
+
returns `WARN (ABSTRACT-ONLY)` instead of upgrading an abstract topic match to
|
|
204
|
+
`ACCEPT`.
|
|
205
|
+
|
|
206
|
+
**Reading a CLI verdict**
|
|
192
207
|
|
|
193
208
|
| Verdict | Meaning |
|
|
194
209
|
|---|---|
|
|
@@ -213,8 +228,9 @@ exit code, so weak or mismatched metadata cannot silently pass automation gates.
|
|
|
213
228
|
|
|
214
229
|
**Full Audit** is for literature search and final pre-submission review. The
|
|
215
230
|
skill fetches abstracts through CrossRef, OpenAlex, Semantic Scholar, Unpaywall,
|
|
216
|
-
arXiv, and PubMed where needed
|
|
217
|
-
claim
|
|
231
|
+
arXiv, and PubMed where needed. For a topline claim, it checks the abstract; for
|
|
232
|
+
a mechanism, implementation, or procedural claim, it continues to a fetched
|
|
233
|
+
full-text passage before assigning support.
|
|
218
234
|
|
|
219
235
|
For a single DOI-backed claim, the CLI can run the abstract check:
|
|
220
236
|
|
|
@@ -255,13 +271,15 @@ Current `check-claim` error codes:
|
|
|
255
271
|
- `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
|
|
256
272
|
- `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
|
|
257
273
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
258
|
-
- `DOI_NOT_FOUND`: selected source did not find a DOI-bound record.
|
|
274
|
+
- `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
|
|
275
|
+
- `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
|
|
259
276
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
260
277
|
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
261
278
|
|
|
262
279
|
> Core rule: every content statement about a paper must come from a live-fetched
|
|
263
|
-
>
|
|
264
|
-
>
|
|
280
|
+
> source at the depth the claim requires — abstract for topline claims, full
|
|
281
|
+
> text for mechanism, implementation, or procedural claims. If the required
|
|
282
|
+
> source is inaccessible, say so. Do not fill the gap from memory.
|
|
265
283
|
|
|
266
284
|
---
|
|
267
285
|
|
|
@@ -298,5 +316,4 @@ that as `WARN (PARTIAL)` instead of accepting the citation.
|
|
|
298
316
|
|
|
299
317
|
## Related
|
|
300
318
|
|
|
301
|
-
- [
|
|
302
|
-
- [decide-skill](https://github.com/Moonweave-Systems/decide-skill) - decision automation for non-expert domains
|
|
319
|
+
- [decision-kernel](https://github.com/Moonweave-Systems/decision-kernel) - evidence-gated decisions and drift/done checks for coding agents
|
|
@@ -215,7 +215,23 @@ _UNSUPPORTED_CLAIM_FRAME_PATTERNS = (
|
|
|
215
215
|
)
|
|
216
216
|
|
|
217
217
|
|
|
218
|
+
def retracted_claim_result(record: PaperRecord, claim: str) -> ClaimSupportResult:
|
|
219
|
+
return ClaimSupportResult(
|
|
220
|
+
status="RETRACTED",
|
|
221
|
+
verdict="REJECT",
|
|
222
|
+
reason=(
|
|
223
|
+
"CrossRef records this paper as retracted "
|
|
224
|
+
f"(notice DOI {record.retraction_doi}); its abstract cannot support the claim."
|
|
225
|
+
),
|
|
226
|
+
evidence="",
|
|
227
|
+
paper=record,
|
|
228
|
+
claim=claim,
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
|
|
218
232
|
def check_claim_support(record: PaperRecord, claim: str) -> ClaimSupportResult:
|
|
233
|
+
if record.retraction_doi:
|
|
234
|
+
return retracted_claim_result(record, claim)
|
|
219
235
|
if not record.abstract:
|
|
220
236
|
return ClaimSupportResult(
|
|
221
237
|
status="UNVERIFIABLE",
|
|
@@ -5,6 +5,7 @@ import json
|
|
|
5
5
|
import sys
|
|
6
6
|
from pathlib import Path
|
|
7
7
|
from typing import Sequence
|
|
8
|
+
from urllib.error import HTTPError
|
|
8
9
|
|
|
9
10
|
from ref_verify.abstract_lookup import (
|
|
10
11
|
AbstractSourceClient,
|
|
@@ -18,7 +19,7 @@ from ref_verify.batch import (
|
|
|
18
19
|
parse_claim_file,
|
|
19
20
|
render_batch_text,
|
|
20
21
|
)
|
|
21
|
-
from ref_verify.claim_check import check_claim_support
|
|
22
|
+
from ref_verify.claim_check import check_claim_support, retracted_claim_result
|
|
22
23
|
from ref_verify.crossref import CrossrefClient
|
|
23
24
|
from ref_verify.doi_check import normalize_doi, verify_doi_metadata
|
|
24
25
|
from ref_verify.models import CitationInput, ClaimSupportResult
|
|
@@ -88,13 +89,31 @@ def _build_parser() -> argparse.ArgumentParser:
|
|
|
88
89
|
|
|
89
90
|
def _verify_doi(args: argparse.Namespace, client: CrossrefClient) -> int:
|
|
90
91
|
lookup_doi = normalize_doi(args.doi)
|
|
91
|
-
fetched = client.fetch_work(lookup_doi)
|
|
92
92
|
provided = CitationInput(
|
|
93
93
|
doi=args.doi,
|
|
94
94
|
title=args.title,
|
|
95
95
|
first_author=args.first_author,
|
|
96
96
|
year=args.year,
|
|
97
97
|
)
|
|
98
|
+
try:
|
|
99
|
+
fetched = client.fetch_work(lookup_doi)
|
|
100
|
+
except HTTPError as exc:
|
|
101
|
+
if not _is_not_found(exc):
|
|
102
|
+
raise
|
|
103
|
+
# A dead DOI is a verdict, not a tool failure; keep the `verdict` key so
|
|
104
|
+
# callers that parse JSON do not have to special-case the error shape.
|
|
105
|
+
_emit(
|
|
106
|
+
{
|
|
107
|
+
"verdict": "REJECT",
|
|
108
|
+
"mismatches": ["doi"],
|
|
109
|
+
"reason": "CrossRef has no record for this DOI (HTTP 404).",
|
|
110
|
+
"provided": provided.to_dict(),
|
|
111
|
+
"fetched": None,
|
|
112
|
+
"error_code": "DOI_NOT_FOUND",
|
|
113
|
+
},
|
|
114
|
+
as_json=args.json,
|
|
115
|
+
)
|
|
116
|
+
return 2
|
|
98
117
|
result = verify_doi_metadata(provided, fetched)
|
|
99
118
|
_emit(result.to_dict(), as_json=args.json)
|
|
100
119
|
return 0 if result.verdict == "PASS" else 2
|
|
@@ -147,7 +166,20 @@ def _run_claim_check(
|
|
|
147
166
|
lookup_doi = normalize_doi(doi)
|
|
148
167
|
selected_clients = _select_abstract_clients(fallback_clients, source)
|
|
149
168
|
if source in ("auto", "crossref"):
|
|
150
|
-
|
|
169
|
+
try:
|
|
170
|
+
fetched = client.fetch_work(lookup_doi)
|
|
171
|
+
except HTTPError as exc:
|
|
172
|
+
if not _is_not_found(exc):
|
|
173
|
+
raise
|
|
174
|
+
return _doi_not_found_payload(claim)
|
|
175
|
+
if fetched.retraction_doi:
|
|
176
|
+
# Fallback abstract sources would drop the retraction flag, so decide here.
|
|
177
|
+
result = retracted_claim_result(fetched, claim)
|
|
178
|
+
payload = result.to_dict()
|
|
179
|
+
payload["abstract_source"] = None
|
|
180
|
+
payload["source_attempts"] = []
|
|
181
|
+
payload["error_code"] = "PAPER_RETRACTED"
|
|
182
|
+
return payload
|
|
151
183
|
lookup_result = lookup_abstract(lookup_doi, fetched, selected_clients)
|
|
152
184
|
else:
|
|
153
185
|
lookup_result = lookup_selected_abstract(lookup_doi, selected_clients)
|
|
@@ -166,6 +198,24 @@ def _run_claim_check(
|
|
|
166
198
|
return _claim_payload(result, lookup_result)
|
|
167
199
|
|
|
168
200
|
|
|
201
|
+
def _is_not_found(exc: HTTPError) -> bool:
|
|
202
|
+
return getattr(exc, "code", None) == 404
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _doi_not_found_payload(claim: str) -> dict:
|
|
206
|
+
return {
|
|
207
|
+
"status": "UNVERIFIABLE",
|
|
208
|
+
"verdict": "REJECT",
|
|
209
|
+
"reason": "CrossRef has no record for this DOI (HTTP 404).",
|
|
210
|
+
"evidence": "",
|
|
211
|
+
"paper": None,
|
|
212
|
+
"claim": claim,
|
|
213
|
+
"abstract_source": None,
|
|
214
|
+
"source_attempts": [],
|
|
215
|
+
"error_code": "DOI_NOT_FOUND",
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
|
|
169
219
|
def _row_error_payload(claim: str, exc: Exception) -> dict:
|
|
170
220
|
return {
|
|
171
221
|
"status": "UNVERIFIABLE",
|
|
@@ -190,6 +240,8 @@ def _claim_payload(result: ClaimSupportResult, lookup_result) -> dict:
|
|
|
190
240
|
def _claim_error_code(result: ClaimSupportResult) -> str:
|
|
191
241
|
if result.verdict == "ACCEPT":
|
|
192
242
|
return "CLAIM_SUPPORTED"
|
|
243
|
+
if result.status == "RETRACTED":
|
|
244
|
+
return "PAPER_RETRACTED"
|
|
193
245
|
if result.status == "UNVERIFIABLE":
|
|
194
246
|
return "NO_ABSTRACT"
|
|
195
247
|
if result.status == "PARTIAL":
|
|
@@ -54,9 +54,24 @@ def parse_crossref_work(message: dict[str, Any]) -> PaperRecord:
|
|
|
54
54
|
source="CrossRef",
|
|
55
55
|
journal=journal,
|
|
56
56
|
url=str(url) if url else None,
|
|
57
|
+
retraction_doi=_retraction_doi(message),
|
|
57
58
|
)
|
|
58
59
|
|
|
59
60
|
|
|
61
|
+
_RETRACTION_UPDATE_TYPES = {"retraction", "withdrawal", "removal", "partial_retraction"}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _retraction_doi(message: dict[str, Any]) -> str | None:
|
|
65
|
+
# CrossRef lists retraction notices under `updated-by`; the retracted work itself
|
|
66
|
+
# keeps its original DOI, so the metadata alone looks like a normal paper.
|
|
67
|
+
for update in message.get("updated-by") or []:
|
|
68
|
+
if not isinstance(update, dict):
|
|
69
|
+
continue
|
|
70
|
+
if str(update.get("type") or "").lower() in _RETRACTION_UPDATE_TYPES:
|
|
71
|
+
return str(update.get("DOI") or "") or "unknown"
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
60
75
|
def _first_string(value: Any) -> str | None:
|
|
61
76
|
if isinstance(value, list) and value:
|
|
62
77
|
return str(value[0]).strip()
|
|
@@ -75,9 +75,18 @@ def verify_doi_metadata(
|
|
|
75
75
|
):
|
|
76
76
|
mismatches.append("year")
|
|
77
77
|
|
|
78
|
+
if fetched.retraction_doi:
|
|
79
|
+
mismatches.append("retracted")
|
|
80
|
+
|
|
78
81
|
if not mismatches:
|
|
79
82
|
verdict = "PASS"
|
|
80
83
|
reason = "Provided citation metadata matches the fetched CrossRef record."
|
|
84
|
+
elif "retracted" in mismatches:
|
|
85
|
+
verdict = "REJECT"
|
|
86
|
+
reason = (
|
|
87
|
+
"CrossRef records this paper as retracted "
|
|
88
|
+
f"(notice DOI {fetched.retraction_doi}); do not use it as a source."
|
|
89
|
+
)
|
|
81
90
|
elif any(field in mismatches for field in ("doi", "title", "first_author")):
|
|
82
91
|
verdict = "REJECT"
|
|
83
92
|
reason = "DOI resolves to a materially different paper than provided."
|
|
@@ -128,12 +137,20 @@ def _strip_trailing_doi_punctuation(value: str) -> str:
|
|
|
128
137
|
|
|
129
138
|
|
|
130
139
|
def _titles_match(provided: str, fetched: str) -> bool:
|
|
140
|
+
provided = _strip_retraction_prefix(provided)
|
|
141
|
+
fetched = _strip_retraction_prefix(fetched)
|
|
131
142
|
if _numbers(provided) != _numbers(fetched):
|
|
132
143
|
return False
|
|
133
144
|
|
|
134
145
|
return _title_tokens(provided) == _title_tokens(fetched)
|
|
135
146
|
|
|
136
147
|
|
|
148
|
+
def _strip_retraction_prefix(title: str) -> str:
|
|
149
|
+
# Publishers prepend "RETRACTED:" to the stored title, and a citation may or may not
|
|
150
|
+
# carry it; the retraction itself is reported separately, never as a title mismatch.
|
|
151
|
+
return re.sub(r"^\s*(?:retracted|withdrawn|retraction)\s*:\s*", "", title, flags=re.IGNORECASE)
|
|
152
|
+
|
|
153
|
+
|
|
137
154
|
def _author_matches(provided: str, fetched: str | None) -> bool:
|
|
138
155
|
if not fetched:
|
|
139
156
|
return False
|
|
@@ -231,6 +231,7 @@ class NumericExpression:
|
|
|
231
231
|
comparator: str
|
|
232
232
|
subject_terms: set[str]
|
|
233
233
|
evidence: str
|
|
234
|
+
prefix_scale: str = ""
|
|
234
235
|
|
|
235
236
|
|
|
236
237
|
def check_numeric_claim_support(abstract: str, claim: str) -> NumericClaimResult:
|
|
@@ -253,7 +254,7 @@ def check_numeric_claim_support(abstract: str, claim: str) -> NumericClaimResult
|
|
|
253
254
|
if _subject_terms_match(claim_expression.subject_terms, subject_context):
|
|
254
255
|
related_evidence = related_evidence or clause
|
|
255
256
|
for evidence_expression in evidence_expressions:
|
|
256
|
-
if not _units_match(claim_expression
|
|
257
|
+
if not _units_match(claim_expression, evidence_expression):
|
|
257
258
|
continue
|
|
258
259
|
if not _subject_terms_match(claim_expression.subject_terms, subject_context):
|
|
259
260
|
continue
|
|
@@ -287,6 +288,7 @@ def _extract_claim_expression(claim: str) -> NumericExpression | None:
|
|
|
287
288
|
comparator=_claim_comparator(claim[: match.start()]),
|
|
288
289
|
subject_terms=_subject_terms(claim[: match.start()]),
|
|
289
290
|
evidence=claim,
|
|
291
|
+
prefix_scale=_prefix_scale(match.group("unit")),
|
|
290
292
|
)
|
|
291
293
|
|
|
292
294
|
|
|
@@ -303,6 +305,7 @@ def _extract_evidence_expressions(clause: str) -> list[NumericExpression]:
|
|
|
303
305
|
),
|
|
304
306
|
subject_terms=_subject_terms(clause[: match.start()]),
|
|
305
307
|
evidence=clause,
|
|
308
|
+
prefix_scale=_prefix_scale(match.group("unit")),
|
|
306
309
|
)
|
|
307
310
|
)
|
|
308
311
|
return expressions
|
|
@@ -412,8 +415,37 @@ def _normalize_unit(value: str) -> str:
|
|
|
412
415
|
return normalized.rstrip("s")
|
|
413
416
|
|
|
414
417
|
|
|
415
|
-
def _units_match(left:
|
|
416
|
-
|
|
418
|
+
def _units_match(left: NumericExpression, right: NumericExpression) -> bool:
|
|
419
|
+
if left.unit != right.unit:
|
|
420
|
+
return False
|
|
421
|
+
# "mV" and "MV" differ by a factor of a million; only an all-lowercase token
|
|
422
|
+
# (common in typed claims) is left ambiguous and allowed to match either.
|
|
423
|
+
if left.prefix_scale and right.prefix_scale:
|
|
424
|
+
return left.prefix_scale == right.prefix_scale
|
|
425
|
+
return True
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
_PREFIX_SENSITIVE_UNITS = {
|
|
429
|
+
"ma",
|
|
430
|
+
"mev",
|
|
431
|
+
"mhz",
|
|
432
|
+
"mohm-cm",
|
|
433
|
+
"mpa",
|
|
434
|
+
"ms/m",
|
|
435
|
+
"mv",
|
|
436
|
+
"mv/cm",
|
|
437
|
+
"mv/m",
|
|
438
|
+
"mv/mm",
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _prefix_scale(raw_unit: str) -> str:
|
|
443
|
+
if _normalize_unit(raw_unit) not in _PREFIX_SENSITIVE_UNITS:
|
|
444
|
+
return ""
|
|
445
|
+
stripped = raw_unit.strip()
|
|
446
|
+
if stripped == stripped.lower():
|
|
447
|
+
return ""
|
|
448
|
+
return "mega" if stripped[0] == "M" else "milli"
|
|
417
449
|
|
|
418
450
|
|
|
419
451
|
def _claim_comparator(prefix: str) -> str:
|
|
@@ -534,6 +566,20 @@ def _evidence_entails_claim(
|
|
|
534
566
|
if claim_comparator == "up_to":
|
|
535
567
|
return evidence_value == claim_value
|
|
536
568
|
return claim_comparator in {"lt", "lte"} and evidence_value <= claim_value
|
|
569
|
+
# Evidence that only bounds the value from one side cannot support a claim
|
|
570
|
+
# bounding it from the other side ("below 50" never establishes "above 40").
|
|
571
|
+
if evidence_comparator in {"lt", "lte"}:
|
|
572
|
+
if claim_comparator == "lt":
|
|
573
|
+
return evidence_value <= claim_value if evidence_comparator == "lt" else evidence_value < claim_value
|
|
574
|
+
if claim_comparator == "lte":
|
|
575
|
+
return evidence_value <= claim_value
|
|
576
|
+
return False
|
|
577
|
+
if evidence_comparator in {"gt", "gte"}:
|
|
578
|
+
if claim_comparator == "gt":
|
|
579
|
+
return evidence_value >= claim_value if evidence_comparator == "gt" else evidence_value > claim_value
|
|
580
|
+
if claim_comparator == "gte":
|
|
581
|
+
return evidence_value >= claim_value
|
|
582
|
+
return False
|
|
537
583
|
if claim_comparator == "gt":
|
|
538
584
|
return evidence_value > claim_value
|
|
539
585
|
if claim_comparator == "gte":
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ref-verify
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: Executable DOI and claim verification helpers for academic citations
|
|
5
5
|
Author: Moonweave Research
|
|
6
6
|
License-Expression: MIT
|
|
@@ -45,6 +45,8 @@ Works with **Claude Code, Cursor, Codex**, and any agent that supports the
|
|
|
45
45
|
|
|
46
46
|
After installation, use it like a normal agent skill. You do not start a server and you do not configure MCP for this workflow. No MCP server is required for this workflow.
|
|
47
47
|
|
|
48
|
+
The skill includes its own copy of the CLI engine, and the agent runs it from the skill folder, so nothing else needs to be installed. Python 3.10 or newer must be available as `python3`.
|
|
49
|
+
|
|
48
50
|
For explicit agent tool-calling rules, see [AGENT_USAGE.md](https://github.com/Moonweave-Research/ref-verify/blob/main/AGENT_USAGE.md).
|
|
49
51
|
|
|
50
52
|
---
|
|
@@ -85,13 +87,20 @@ checks that are currently safe to automate directly:
|
|
|
85
87
|
- JSON output for agent-readable routing
|
|
86
88
|
- Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
|
|
87
89
|
|
|
88
|
-
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol.
|
|
90
|
+
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol. The CLI rejects a DOI that CrossRef records as retracted (via its retraction notice); retraction banners CrossRef does not know about, Unpaywall, arXiv, and two-source existence checks remain in the `SKILL.md` protocol.
|
|
89
91
|
|
|
90
92
|
The CLI has zero third-party Python runtime dependencies, but it is not an
|
|
91
93
|
offline verifier. Functional checks require outbound HTTPS access to public
|
|
92
94
|
academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
|
|
93
95
|
|
|
94
|
-
|
|
96
|
+
To run the CLI yourself, install it from PyPI:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
uvx ref-verify --help # run without installing (uv)
|
|
100
|
+
pipx install ref-verify # or install the `ref-verify` command
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Or install it from a local checkout:
|
|
95
104
|
|
|
96
105
|
```bash
|
|
97
106
|
git clone https://github.com/Moonweave-Research/ref-verify.git
|
|
@@ -172,20 +181,20 @@ normal CI does not fail because an upstream API is temporarily unavailable.
|
|
|
172
181
|
|
|
173
182
|
---
|
|
174
183
|
|
|
175
|
-
## Scope —
|
|
184
|
+
## Scope — optional CLI versus manual audit
|
|
176
185
|
|
|
177
186
|
`ref-verify` is a conservative guard, not an oracle. It errs toward flagging: an
|
|
178
187
|
`ACCEPT` is high-confidence, and **anything else means "not auto-verifiable —
|
|
179
188
|
check it yourself", not "the citation is wrong."**
|
|
180
189
|
|
|
181
|
-
**
|
|
190
|
+
**The optional CLI verifies**
|
|
182
191
|
|
|
183
192
|
- DOI metadata: title, first-author surname, and year against CrossRef.
|
|
184
193
|
- Whether a DOI-bound **abstract** explicitly supports a specific numeric or
|
|
185
194
|
literal claim, quoted verbatim. If no abstract is reachable, it returns
|
|
186
195
|
`UNVERIFIABLE` rather than guessing.
|
|
187
196
|
|
|
188
|
-
**
|
|
197
|
+
**The optional CLI does not verify** (out of scope by design, not bugs)
|
|
189
198
|
|
|
190
199
|
- **Full-text, figure, table, or supplementary values** — abstract-only. A number
|
|
191
200
|
that appears only in the body stays `UNVERIFIABLE`.
|
|
@@ -199,7 +208,13 @@ check it yourself", not "the citation is wrong."**
|
|
|
199
208
|
- **Paper quality, novelty, field consensus**, or whether the *full* paper
|
|
200
209
|
supports a broader statement.
|
|
201
210
|
|
|
202
|
-
|
|
211
|
+
The agent skill's manual Full Audit protocol goes beyond the optional CLI for
|
|
212
|
+
mechanism, implementation, and procedural claims. It requires a fetched
|
|
213
|
+
full-text passage at that source depth; when full text is unavailable, it
|
|
214
|
+
returns `WARN (ABSTRACT-ONLY)` instead of upgrading an abstract topic match to
|
|
215
|
+
`ACCEPT`.
|
|
216
|
+
|
|
217
|
+
**Reading a CLI verdict**
|
|
203
218
|
|
|
204
219
|
| Verdict | Meaning |
|
|
205
220
|
|---|---|
|
|
@@ -224,8 +239,9 @@ exit code, so weak or mismatched metadata cannot silently pass automation gates.
|
|
|
224
239
|
|
|
225
240
|
**Full Audit** is for literature search and final pre-submission review. The
|
|
226
241
|
skill fetches abstracts through CrossRef, OpenAlex, Semantic Scholar, Unpaywall,
|
|
227
|
-
arXiv, and PubMed where needed
|
|
228
|
-
claim
|
|
242
|
+
arXiv, and PubMed where needed. For a topline claim, it checks the abstract; for
|
|
243
|
+
a mechanism, implementation, or procedural claim, it continues to a fetched
|
|
244
|
+
full-text passage before assigning support.
|
|
229
245
|
|
|
230
246
|
For a single DOI-backed claim, the CLI can run the abstract check:
|
|
231
247
|
|
|
@@ -266,13 +282,15 @@ Current `check-claim` error codes:
|
|
|
266
282
|
- `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
|
|
267
283
|
- `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
|
|
268
284
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
269
|
-
- `DOI_NOT_FOUND`: selected source did not find a DOI-bound record.
|
|
285
|
+
- `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
|
|
286
|
+
- `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
|
|
270
287
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
271
288
|
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
272
289
|
|
|
273
290
|
> Core rule: every content statement about a paper must come from a live-fetched
|
|
274
|
-
>
|
|
275
|
-
>
|
|
291
|
+
> source at the depth the claim requires — abstract for topline claims, full
|
|
292
|
+
> text for mechanism, implementation, or procedural claims. If the required
|
|
293
|
+
> source is inaccessible, say so. Do not fill the gap from memory.
|
|
276
294
|
|
|
277
295
|
---
|
|
278
296
|
|
|
@@ -309,5 +327,4 @@ that as `WARN (PARTIAL)` instead of accepting the citation.
|
|
|
309
327
|
|
|
310
328
|
## Related
|
|
311
329
|
|
|
312
|
-
- [
|
|
313
|
-
- [decide-skill](https://github.com/Moonweave-Systems/decide-skill) - decision automation for non-expert domains
|
|
330
|
+
- [decision-kernel](https://github.com/Moonweave-Systems/decision-kernel) - evidence-gated decisions and drift/done checks for coding agents
|
|
@@ -4,6 +4,7 @@ import tempfile
|
|
|
4
4
|
import unittest
|
|
5
5
|
from contextlib import redirect_stdout
|
|
6
6
|
from pathlib import Path
|
|
7
|
+
from urllib.error import HTTPError
|
|
7
8
|
|
|
8
9
|
from ref_verify.abstract_lookup import AbstractSourceError
|
|
9
10
|
from ref_verify.cli import _default_abstract_clients, main
|
|
@@ -302,6 +303,77 @@ class CliTests(unittest.TestCase):
|
|
|
302
303
|
self.assertEqual(payload["abstract_source"], "crossref")
|
|
303
304
|
self.assertEqual(payload["error_code"], "CLAIM_SUPPORTED")
|
|
304
305
|
|
|
306
|
+
def test_verify_doi_reports_dead_doi_as_reject_with_verdict_key(self):
|
|
307
|
+
error = HTTPError("https://api.crossref.org/works/x", 404, "Not Found", None, None)
|
|
308
|
+
output = io.StringIO()
|
|
309
|
+
|
|
310
|
+
with redirect_stdout(output):
|
|
311
|
+
exit_code = main(
|
|
312
|
+
["verify-doi", "10.1126/science.999999.9999", "--json"],
|
|
313
|
+
client=FailingClient(error),
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
payload = json.loads(output.getvalue())
|
|
317
|
+
self.assertEqual(exit_code, 2)
|
|
318
|
+
self.assertEqual(payload["verdict"], "REJECT")
|
|
319
|
+
self.assertEqual(payload["error_code"], "DOI_NOT_FOUND")
|
|
320
|
+
self.assertNotIn("error", payload)
|
|
321
|
+
|
|
322
|
+
def test_verify_doi_still_surfaces_non_404_http_errors(self):
|
|
323
|
+
error = HTTPError("https://api.crossref.org/works/x", 503, "Unavailable", None, None)
|
|
324
|
+
output = io.StringIO()
|
|
325
|
+
|
|
326
|
+
with redirect_stdout(output):
|
|
327
|
+
exit_code = main(
|
|
328
|
+
["verify-doi", "10.1000/example", "--json"],
|
|
329
|
+
client=FailingClient(error),
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
payload = json.loads(output.getvalue())
|
|
333
|
+
self.assertEqual(exit_code, 1)
|
|
334
|
+
self.assertIn("error", payload)
|
|
335
|
+
|
|
336
|
+
def test_check_claim_reports_dead_doi_as_reject(self):
|
|
337
|
+
error = HTTPError("https://api.crossref.org/works/x", 404, "Not Found", None, None)
|
|
338
|
+
output = io.StringIO()
|
|
339
|
+
|
|
340
|
+
with redirect_stdout(output):
|
|
341
|
+
exit_code = main(
|
|
342
|
+
["check-claim", "10.1126/science.999999.9999", "--claim", "strain above 100%", "--json"],
|
|
343
|
+
client=FailingClient(error),
|
|
344
|
+
abstract_clients=[],
|
|
345
|
+
)
|
|
346
|
+
|
|
347
|
+
payload = json.loads(output.getvalue())
|
|
348
|
+
self.assertEqual(exit_code, 2)
|
|
349
|
+
self.assertEqual(payload["verdict"], "REJECT")
|
|
350
|
+
self.assertEqual(payload["error_code"], "DOI_NOT_FOUND")
|
|
351
|
+
|
|
352
|
+
def test_check_claim_rejects_retracted_paper_before_reading_abstract(self):
|
|
353
|
+
record = PaperRecord(
|
|
354
|
+
doi="10.1000/retracted",
|
|
355
|
+
title="RETRACTED: Dielectric elastomer actuators",
|
|
356
|
+
authors=["Pelrine"],
|
|
357
|
+
year=2000,
|
|
358
|
+
abstract="Actuated strains up to 117% were demonstrated.",
|
|
359
|
+
source="fixture",
|
|
360
|
+
retraction_doi="10.1000/retraction-notice",
|
|
361
|
+
)
|
|
362
|
+
output = io.StringIO()
|
|
363
|
+
|
|
364
|
+
with redirect_stdout(output):
|
|
365
|
+
exit_code = main(
|
|
366
|
+
["check-claim", "10.1000/retracted", "--claim", "actuation strain above 100%", "--json"],
|
|
367
|
+
client=FakeClient(record),
|
|
368
|
+
abstract_clients=[],
|
|
369
|
+
)
|
|
370
|
+
|
|
371
|
+
payload = json.loads(output.getvalue())
|
|
372
|
+
self.assertEqual(exit_code, 2)
|
|
373
|
+
self.assertEqual(payload["status"], "RETRACTED")
|
|
374
|
+
self.assertEqual(payload["verdict"], "REJECT")
|
|
375
|
+
self.assertEqual(payload["error_code"], "PAPER_RETRACTED")
|
|
376
|
+
|
|
305
377
|
def test_check_claim_does_not_query_fallback_when_crossref_has_abstract(self):
|
|
306
378
|
record = PaperRecord(
|
|
307
379
|
doi="10.1000/example",
|
|
@@ -41,6 +41,33 @@ class CrossrefTests(unittest.TestCase):
|
|
|
41
41
|
|
|
42
42
|
self.assertIsNone(record.year)
|
|
43
43
|
|
|
44
|
+
def test_reads_retraction_notice_from_updated_by(self):
|
|
45
|
+
message = {
|
|
46
|
+
"DOI": "10.1016/s0140-6736(97)11096-0",
|
|
47
|
+
"title": ["RETRACTED: Ileal-lymphoid-nodular hyperplasia"],
|
|
48
|
+
"author": [{"family": "Wakefield", "given": "AJ"}],
|
|
49
|
+
"issued": {"date-parts": [[1998]]},
|
|
50
|
+
"updated-by": [
|
|
51
|
+
{"type": "correction", "DOI": "10.1016/s0140-6736(04)15715-2"},
|
|
52
|
+
{"type": "retraction", "DOI": "10.1016/s0140-6736(10)60175-4"},
|
|
53
|
+
],
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
record = parse_crossref_work(message)
|
|
57
|
+
|
|
58
|
+
self.assertEqual(record.retraction_doi, "10.1016/s0140-6736(10)60175-4")
|
|
59
|
+
|
|
60
|
+
def test_correction_alone_is_not_a_retraction(self):
|
|
61
|
+
message = {
|
|
62
|
+
"DOI": "10.1000/corrected",
|
|
63
|
+
"title": ["Corrected paper"],
|
|
64
|
+
"updated-by": [{"type": "correction", "DOI": "10.1000/corrigendum"}],
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
record = parse_crossref_work(message)
|
|
68
|
+
|
|
69
|
+
self.assertIsNone(record.retraction_doi)
|
|
70
|
+
|
|
44
71
|
def test_preserves_group_author_name(self):
|
|
45
72
|
message = {
|
|
46
73
|
"DOI": "10.1000/group-author",
|
|
@@ -295,6 +295,71 @@ class DoiCheckTests(unittest.TestCase):
|
|
|
295
295
|
self.assertIn("metadata", result.mismatches)
|
|
296
296
|
self.assertIn("year", result.mismatches)
|
|
297
297
|
|
|
298
|
+
def test_rejects_retracted_paper_even_when_metadata_matches(self):
|
|
299
|
+
provided = CitationInput(
|
|
300
|
+
doi="10.1000/retracted",
|
|
301
|
+
title="Ileal-lymphoid-nodular hyperplasia",
|
|
302
|
+
first_author="Wakefield",
|
|
303
|
+
year=1998,
|
|
304
|
+
)
|
|
305
|
+
fetched = PaperRecord(
|
|
306
|
+
doi="10.1000/retracted",
|
|
307
|
+
title="RETRACTED: Ileal-lymphoid-nodular hyperplasia",
|
|
308
|
+
authors=["Wakefield"],
|
|
309
|
+
year=1998,
|
|
310
|
+
abstract=None,
|
|
311
|
+
source="fixture",
|
|
312
|
+
retraction_doi="10.1000/retraction-notice",
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
result = verify_doi_metadata(provided, fetched)
|
|
316
|
+
|
|
317
|
+
self.assertEqual(result.verdict, "REJECT")
|
|
318
|
+
self.assertEqual(result.mismatches, ["retracted"])
|
|
319
|
+
self.assertIn("10.1000/retraction-notice", result.reason)
|
|
320
|
+
|
|
321
|
+
def test_retracted_title_prefix_does_not_count_as_a_title_mismatch(self):
|
|
322
|
+
provided = CitationInput(
|
|
323
|
+
doi="10.1000/retracted",
|
|
324
|
+
title="Ileal-lymphoid-nodular hyperplasia",
|
|
325
|
+
first_author="Wakefield",
|
|
326
|
+
year=1998,
|
|
327
|
+
)
|
|
328
|
+
fetched = PaperRecord(
|
|
329
|
+
doi="10.1000/retracted",
|
|
330
|
+
title="RETRACTED: Ileal-lymphoid-nodular hyperplasia",
|
|
331
|
+
authors=["Wakefield"],
|
|
332
|
+
year=1998,
|
|
333
|
+
abstract=None,
|
|
334
|
+
source="fixture",
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
result = verify_doi_metadata(provided, fetched)
|
|
338
|
+
|
|
339
|
+
self.assertNotIn("title", result.mismatches)
|
|
340
|
+
|
|
341
|
+
def test_citation_carrying_the_retracted_prefix_is_not_a_title_mismatch(self):
|
|
342
|
+
provided = CitationInput(
|
|
343
|
+
doi="10.1000/retracted",
|
|
344
|
+
title="RETRACTED: Ileal-lymphoid-nodular hyperplasia",
|
|
345
|
+
first_author="Wakefield",
|
|
346
|
+
year=1998,
|
|
347
|
+
)
|
|
348
|
+
fetched = PaperRecord(
|
|
349
|
+
doi="10.1000/retracted",
|
|
350
|
+
title="RETRACTED: Ileal-lymphoid-nodular hyperplasia",
|
|
351
|
+
authors=["Wakefield"],
|
|
352
|
+
year=1998,
|
|
353
|
+
abstract=None,
|
|
354
|
+
source="fixture",
|
|
355
|
+
retraction_doi="10.1000/retraction-notice",
|
|
356
|
+
)
|
|
357
|
+
|
|
358
|
+
result = verify_doi_metadata(provided, fetched)
|
|
359
|
+
|
|
360
|
+
self.assertEqual(result.verdict, "REJECT")
|
|
361
|
+
self.assertEqual(result.mismatches, ["retracted"])
|
|
362
|
+
|
|
298
363
|
|
|
299
364
|
if __name__ == "__main__":
|
|
300
365
|
unittest.main()
|
|
@@ -229,6 +229,46 @@ class NumericClaimTests(unittest.TestCase):
|
|
|
229
229
|
|
|
230
230
|
self.assertEqual(result.status, "PARTIAL")
|
|
231
231
|
|
|
232
|
+
def test_upper_bounded_evidence_does_not_support_lower_bounded_claim(self):
|
|
233
|
+
for abstract, claim in (
|
|
234
|
+
("The modulus was below 50 MPa.", "modulus above 40 MPa"),
|
|
235
|
+
("The breakdown field was at most 100 MV/m.", "breakdown field at least 80 MV/m"),
|
|
236
|
+
("The modulus was at least 50 MPa.", "modulus below 60 MPa"),
|
|
237
|
+
):
|
|
238
|
+
with self.subTest(abstract=abstract, claim=claim):
|
|
239
|
+
result = check_numeric_claim_support(abstract, claim)
|
|
240
|
+
|
|
241
|
+
self.assertEqual(result.status, "PARTIAL")
|
|
242
|
+
|
|
243
|
+
def test_one_sided_evidence_supports_claim_bounded_on_the_same_side(self):
|
|
244
|
+
for abstract, claim in (
|
|
245
|
+
("The modulus was below 50 MPa.", "modulus below 60 MPa"),
|
|
246
|
+
("The modulus was at least 50 MPa.", "modulus above 40 MPa"),
|
|
247
|
+
):
|
|
248
|
+
with self.subTest(abstract=abstract, claim=claim):
|
|
249
|
+
result = check_numeric_claim_support(abstract, claim)
|
|
250
|
+
|
|
251
|
+
self.assertEqual(result.status, "SUPPORTED")
|
|
252
|
+
|
|
253
|
+
def test_milli_and_mega_prefixes_do_not_match(self):
|
|
254
|
+
for abstract, claim in (
|
|
255
|
+
("The output voltage was 5 mV.", "output voltage of 5 MV"),
|
|
256
|
+
("The current reached 2 mA.", "current reached 2 MA"),
|
|
257
|
+
("The modulus was 3 MPa.", "modulus of 3 mPa"),
|
|
258
|
+
):
|
|
259
|
+
with self.subTest(abstract=abstract, claim=claim):
|
|
260
|
+
result = check_numeric_claim_support(abstract, claim)
|
|
261
|
+
|
|
262
|
+
self.assertEqual(result.status, "PARTIAL")
|
|
263
|
+
|
|
264
|
+
def test_lowercase_unit_prefix_stays_ambiguous(self):
|
|
265
|
+
result = check_numeric_claim_support(
|
|
266
|
+
"The output voltage was 5 mV.",
|
|
267
|
+
"output voltage of 5 mv",
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
self.assertEqual(result.status, "SUPPORTED")
|
|
271
|
+
|
|
232
272
|
|
|
233
273
|
class NumericClaimEvalFixtureTests(unittest.TestCase):
|
|
234
274
|
def test_numeric_claim_eval_fixture(self):
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
from pathlib import Path
|
|
2
2
|
import os
|
|
3
3
|
import subprocess
|
|
4
|
+
import tempfile
|
|
4
5
|
import sys
|
|
5
6
|
import unittest
|
|
6
7
|
|
|
@@ -24,6 +25,7 @@ CHECK_CLAIM_ERROR_CODES = (
|
|
|
24
25
|
"SOURCE_TIMEOUT",
|
|
25
26
|
"SOURCE_RATE_LIMITED",
|
|
26
27
|
"SOURCE_UNSUPPORTED",
|
|
28
|
+
"PAPER_RETRACTED",
|
|
27
29
|
)
|
|
28
30
|
AGENT_USAGE_ERROR_CODES = CHECK_CLAIM_ERROR_CODES + ("ROW_CHECK_ERROR",)
|
|
29
31
|
|
|
@@ -74,6 +76,28 @@ class SkillDocsTests(unittest.TestCase):
|
|
|
74
76
|
with self.subTest(error_code=code):
|
|
75
77
|
self.assertIn(code, skill)
|
|
76
78
|
|
|
79
|
+
def test_skill_requires_source_depth_for_mechanism_claims(self):
|
|
80
|
+
skill = (REPO_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
|
81
|
+
readme = (REPO_ROOT / "README.md").read_text(encoding="utf-8")
|
|
82
|
+
readme_ko = (REPO_ROOT / "README.ko.md").read_text(encoding="utf-8")
|
|
83
|
+
|
|
84
|
+
required_skill_phrases = (
|
|
85
|
+
"mechanism/implementation/procedural claims",
|
|
86
|
+
"CLI `ACCEPT` is abstract-level evidence only",
|
|
87
|
+
"Full-Text Confirmation",
|
|
88
|
+
"content supported at the required source depth",
|
|
89
|
+
"fetched source text at the required depth",
|
|
90
|
+
"CONTENT: ABSTRACT-ONLY",
|
|
91
|
+
)
|
|
92
|
+
for phrase in required_skill_phrases:
|
|
93
|
+
with self.subTest(phrase=phrase):
|
|
94
|
+
self.assertIn(phrase, skill)
|
|
95
|
+
|
|
96
|
+
self.assertNotIn("/Users/", skill)
|
|
97
|
+
self.assertNotIn("verify@ref-verify.local", skill)
|
|
98
|
+
self.assertIn("manual Full Audit protocol", readme)
|
|
99
|
+
self.assertIn("수동 Full Audit 프로토콜", readme_ko)
|
|
100
|
+
|
|
77
101
|
def test_readme_positions_cli_as_skill_execution_engine_not_mcp(self):
|
|
78
102
|
readme = (REPO_ROOT / "README.md").read_text(encoding="utf-8")
|
|
79
103
|
|
|
@@ -204,8 +228,34 @@ class SkillDocsTests(unittest.TestCase):
|
|
|
204
228
|
encoding="utf-8"
|
|
205
229
|
)
|
|
206
230
|
|
|
207
|
-
self.assertIn('version = "1.2.
|
|
208
|
-
self.assertIn('__version__ = "1.2.
|
|
231
|
+
self.assertIn('version = "1.2.2"', pyproject)
|
|
232
|
+
self.assertIn('__version__ = "1.2.2"', init)
|
|
233
|
+
|
|
234
|
+
def test_skill_runs_bundled_engine_by_absolute_path(self):
|
|
235
|
+
# npx skills add copies src/ next to SKILL.md but installs no console script, and the
|
|
236
|
+
# agent's working directory is the user's project, so a relative PYTHONPATH=src fails.
|
|
237
|
+
skill = (REPO_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
|
238
|
+
|
|
239
|
+
self.assertNotIn("PYTHONPATH=src ", skill)
|
|
240
|
+
self.assertIn('PYTHONPATH="$SKILL_DIR/src" python3 -m ref_verify.cli --help', skill)
|
|
241
|
+
self.assertIn('PYTHONPATH="$SKILL_DIR/src" python3 -m ref_verify.cli verify-doi', skill)
|
|
242
|
+
self.assertIn('PYTHONPATH="$SKILL_DIR/src" python3 -m ref_verify.cli check-claim', skill)
|
|
243
|
+
self.assertInOrder(skill, ("$SKILL_DIR/src", "ref-verify --help", "uvx --from 'ref-verify>=1.2.1'"))
|
|
244
|
+
|
|
245
|
+
def test_bundled_engine_runs_from_another_working_directory(self):
|
|
246
|
+
env = os.environ.copy()
|
|
247
|
+
env["PYTHONPATH"] = str(REPO_ROOT / "src")
|
|
248
|
+
with tempfile.TemporaryDirectory() as project:
|
|
249
|
+
result = subprocess.run(
|
|
250
|
+
[sys.executable, "-m", "ref_verify.cli", "--help"],
|
|
251
|
+
cwd=project,
|
|
252
|
+
env=env,
|
|
253
|
+
capture_output=True,
|
|
254
|
+
text=True,
|
|
255
|
+
check=False,
|
|
256
|
+
)
|
|
257
|
+
self.assertEqual(result.returncode, 0, result.stderr)
|
|
258
|
+
self.assertIn("verify-doi", result.stdout)
|
|
209
259
|
|
|
210
260
|
def test_readmes_prioritize_user_workflow_before_architecture_details(self):
|
|
211
261
|
readme = (REPO_ROOT / "README.md").read_text(encoding="utf-8")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|