ref-verify 1.1.2__tar.gz → 1.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ref_verify-1.1.2/src/ref_verify.egg-info → ref_verify-1.2.2}/PKG-INFO +95 -15
- {ref_verify-1.1.2 → ref_verify-1.2.2}/README.md +94 -14
- {ref_verify-1.1.2 → ref_verify-1.2.2}/pyproject.toml +1 -1
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/__init__.py +1 -1
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/abstract_lookup.py +2 -0
- ref_verify-1.2.2/src/ref_verify/batch.py +239 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/claim_check.py +21 -1
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/cli.py +132 -14
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/crossref.py +15 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/doi_check.py +17 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/models.py +1 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/numeric_claim.py +182 -16
- ref_verify-1.2.2/src/ref_verify/openalex.py +123 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/semantic_scholar.py +41 -14
- {ref_verify-1.1.2 → ref_verify-1.2.2/src/ref_verify.egg-info}/PKG-INFO +95 -15
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify.egg-info/SOURCES.txt +3 -0
- ref_verify-1.2.2/tests/test_abstract_sources.py +234 -0
- ref_verify-1.2.2/tests/test_batch.py +238 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/tests/test_claim_check.py +36 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/tests/test_cli.py +495 -1
- {ref_verify-1.1.2 → ref_verify-1.2.2}/tests/test_crossref.py +27 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/tests/test_doi_check.py +65 -0
- ref_verify-1.2.2/tests/test_numeric_claim.py +292 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/tests/test_package_smoke.py +6 -2
- {ref_verify-1.1.2 → ref_verify-1.2.2}/tests/test_skill_docs.py +89 -7
- ref_verify-1.1.2/tests/test_abstract_sources.py +0 -105
- ref_verify-1.1.2/tests/test_numeric_claim.py +0 -88
- {ref_verify-1.1.2 → ref_verify-1.2.2}/LICENSE +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/setup.cfg +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify/pubmed.py +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify.egg-info/dependency_links.txt +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify.egg-info/entry_points.txt +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.2}/src/ref_verify.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ref-verify
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: Executable DOI and claim verification helpers for academic citations
|
|
5
5
|
Author: Moonweave Research
|
|
6
6
|
License-Expression: MIT
|
|
@@ -45,6 +45,10 @@ Works with **Claude Code, Cursor, Codex**, and any agent that supports the
|
|
|
45
45
|
|
|
46
46
|
After installation, use it like a normal agent skill. You do not start a server and you do not configure MCP for this workflow. No MCP server is required for this workflow.
|
|
47
47
|
|
|
48
|
+
The skill includes its own copy of the CLI engine, and the agent runs it from the skill folder, so nothing else needs to be installed. Python 3.10 or newer must be available as `python3`.
|
|
49
|
+
|
|
50
|
+
For explicit agent tool-calling rules, see [AGENT_USAGE.md](https://github.com/Moonweave-Research/ref-verify/blob/main/AGENT_USAGE.md).
|
|
51
|
+
|
|
48
52
|
---
|
|
49
53
|
|
|
50
54
|
## Use it
|
|
@@ -75,20 +79,28 @@ checks that are currently safe to automate directly:
|
|
|
75
79
|
|
|
76
80
|
- CrossRef metadata check: `ref-verify verify-doi`
|
|
77
81
|
- DOI-bound abstract claim check: `ref-verify check-claim`
|
|
82
|
+
- Batch DOI-bound claim checks: `ref-verify check-file`
|
|
78
83
|
- literal text claims
|
|
79
84
|
- subject-matched percentage claims such as efficiency, response rate, or actuation strain
|
|
80
85
|
- simple unit/count claims such as cycles, patients, voltage, temperature, and concentration
|
|
81
|
-
- CrossRef first, then DOI-bound Semantic Scholar and PubMed fallback when CrossRef has no abstract
|
|
86
|
+
- CrossRef first, then DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback when CrossRef has no abstract
|
|
82
87
|
- JSON output for agent-readable routing
|
|
83
88
|
- Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
|
|
84
89
|
|
|
85
|
-
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol.
|
|
90
|
+
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol. The CLI rejects a DOI that CrossRef records as retracted (via its retraction notice); retraction banners CrossRef does not know about, Unpaywall, arXiv, and two-source existence checks remain in the `SKILL.md` protocol.
|
|
86
91
|
|
|
87
92
|
The CLI has zero third-party Python runtime dependencies, but it is not an
|
|
88
93
|
offline verifier. Functional checks require outbound HTTPS access to public
|
|
89
|
-
academic APIs such as CrossRef, Semantic Scholar, and PubMed.
|
|
94
|
+
academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
|
|
90
95
|
|
|
91
|
-
|
|
96
|
+
To run the CLI yourself, install it from PyPI:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
uvx ref-verify --help # run without installing (uv)
|
|
100
|
+
pipx install ref-verify # or install the `ref-verify` command
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Or install it from a local checkout:
|
|
92
104
|
|
|
93
105
|
```bash
|
|
94
106
|
git clone https://github.com/Moonweave-Research/ref-verify.git
|
|
@@ -127,7 +139,7 @@ ref-verify check-claim 10.1126/science.287.5454.836 \
|
|
|
127
139
|
--json
|
|
128
140
|
```
|
|
129
141
|
|
|
130
|
-
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound Semantic Scholar and PubMed fallback sources. Use `--source crossref`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
142
|
+
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback sources. Use `--source crossref`, `--source openalex`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
131
143
|
|
|
132
144
|
Source-checkout equivalents:
|
|
133
145
|
|
|
@@ -169,6 +181,50 @@ normal CI does not fail because an upstream API is temporarily unavailable.
|
|
|
169
181
|
|
|
170
182
|
---
|
|
171
183
|
|
|
184
|
+
## Scope — optional CLI versus manual audit
|
|
185
|
+
|
|
186
|
+
`ref-verify` is a conservative guard, not an oracle. It errs toward flagging: an
|
|
187
|
+
`ACCEPT` is high-confidence, and **anything else means "not auto-verifiable —
|
|
188
|
+
check it yourself", not "the citation is wrong."**
|
|
189
|
+
|
|
190
|
+
**The optional CLI verifies**
|
|
191
|
+
|
|
192
|
+
- DOI metadata: title, first-author surname, and year against CrossRef.
|
|
193
|
+
- Whether a DOI-bound **abstract** explicitly supports a specific numeric or
|
|
194
|
+
literal claim, quoted verbatim. If no abstract is reachable, it returns
|
|
195
|
+
`UNVERIFIABLE` rather than guessing.
|
|
196
|
+
|
|
197
|
+
**The optional CLI does not verify** (out of scope by design, not bugs)
|
|
198
|
+
|
|
199
|
+
- **Full-text, figure, table, or supplementary values** — abstract-only. A number
|
|
200
|
+
that appears only in the body stays `UNVERIFIABLE`.
|
|
201
|
+
- **Relational or qualitative claims** — proportionalities, mechanisms,
|
|
202
|
+
"broader/stronger than". Only value+unit and literal claims are checked.
|
|
203
|
+
- **Papers whose publisher withholds the abstract** — some titles expose no
|
|
204
|
+
abstract to CrossRef or OpenAlex. No abstract → `UNVERIFIABLE`, which reflects
|
|
205
|
+
reachability, not the claim.
|
|
206
|
+
- **Statistical metrics** (p-value, AUC/AUROC, F1, hazard/odds ratio, confidence
|
|
207
|
+
intervals) — handled by the manual skill protocol, not the CLI.
|
|
208
|
+
- **Paper quality, novelty, field consensus**, or whether the *full* paper
|
|
209
|
+
supports a broader statement.
|
|
210
|
+
|
|
211
|
+
The agent skill's manual Full Audit protocol goes beyond the optional CLI for
|
|
212
|
+
mechanism, implementation, and procedural claims. It requires a fetched
|
|
213
|
+
full-text passage at that source depth; when full text is unavailable, it
|
|
214
|
+
returns `WARN (ABSTRACT-ONLY)` instead of upgrading an abstract topic match to
|
|
215
|
+
`ACCEPT`.
|
|
216
|
+
|
|
217
|
+
**Reading a CLI verdict**
|
|
218
|
+
|
|
219
|
+
| Verdict | Meaning |
|
|
220
|
+
|---|---|
|
|
221
|
+
| `ACCEPT` | The fetched abstract explicitly supports the claim. High-confidence pass. |
|
|
222
|
+
| `WARN` / `PARTIAL` | An abstract was read but does not explicitly support the exact claim. Check the source. |
|
|
223
|
+
| `UNVERIFIABLE` | No abstract was reachable to check against. Not a judgment on the claim. |
|
|
224
|
+
| `REJECT` | DOI is dead, resolves to a different paper, contradicted, or retracted. |
|
|
225
|
+
|
|
226
|
+
---
|
|
227
|
+
|
|
172
228
|
## Modes
|
|
173
229
|
|
|
174
230
|
**Quick Screen** is for DOIs you already have. It uses CrossRef to compare the
|
|
@@ -182,9 +238,10 @@ ref-verify verify-doi <doi> --title "<title>" --first-author <last-name> --year
|
|
|
182
238
|
exit code, so weak or mismatched metadata cannot silently pass automation gates.
|
|
183
239
|
|
|
184
240
|
**Full Audit** is for literature search and final pre-submission review. The
|
|
185
|
-
skill fetches abstracts through CrossRef, Semantic Scholar, Unpaywall,
|
|
186
|
-
and PubMed where needed
|
|
187
|
-
claim
|
|
241
|
+
skill fetches abstracts through CrossRef, OpenAlex, Semantic Scholar, Unpaywall,
|
|
242
|
+
arXiv, and PubMed where needed. For a topline claim, it checks the abstract; for
|
|
243
|
+
a mechanism, implementation, or procedural claim, it continues to a fetched
|
|
244
|
+
full-text passage before assigning support.
|
|
188
245
|
|
|
189
246
|
For a single DOI-backed claim, the CLI can run the abstract check:
|
|
190
247
|
|
|
@@ -197,19 +254,43 @@ ref-verify check-claim <doi> --claim "<specific claim>" --json
|
|
|
197
254
|
`abstract_source`, `source_attempts`, and `error_code` so agents can distinguish
|
|
198
255
|
missing abstracts, source failures, DOI mismatches, and ambiguous evidence.
|
|
199
256
|
|
|
257
|
+
Use `check-file` when a draft, literature note, or AI-agent output has many
|
|
258
|
+
DOI/claim pairs.
|
|
259
|
+
|
|
260
|
+
JSONL:
|
|
261
|
+
|
|
262
|
+
```bash
|
|
263
|
+
ref-verify check-file claims.jsonl
|
|
264
|
+
ref-verify check-file claims.jsonl --json
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
CSV:
|
|
268
|
+
|
|
269
|
+
```bash
|
|
270
|
+
ref-verify check-file claims.csv
|
|
271
|
+
```
|
|
272
|
+
|
|
273
|
+
Each row must include `doi` and `claim`. Optional fields are `id`, `source`,
|
|
274
|
+
and `note`. Batch mode reuses the same conservative `check-claim` engine:
|
|
275
|
+
`ACCEPT` means the abstract explicitly supports the numeric claim. `WARN`,
|
|
276
|
+
`PARTIAL`, `REJECT`, or `UNVERIFIABLE` means the claim should not be treated as
|
|
277
|
+
verified.
|
|
278
|
+
|
|
200
279
|
Current `check-claim` error codes:
|
|
201
280
|
|
|
202
281
|
- `CLAIM_SUPPORTED`: explicit abstract support found.
|
|
203
282
|
- `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
|
|
204
283
|
- `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
|
|
205
284
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
206
|
-
- `DOI_NOT_FOUND`: selected source did not find a DOI-bound record.
|
|
285
|
+
- `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
|
|
286
|
+
- `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
|
|
207
287
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
208
|
-
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_UNSUPPORTED`: source lookup failed or could not be used.
|
|
288
|
+
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
209
289
|
|
|
210
290
|
> Core rule: every content statement about a paper must come from a live-fetched
|
|
211
|
-
>
|
|
212
|
-
>
|
|
291
|
+
> source at the depth the claim requires — abstract for topline claims, full
|
|
292
|
+
> text for mechanism, implementation, or procedural claims. If the required
|
|
293
|
+
> source is inaccessible, say so. Do not fill the gap from memory.
|
|
213
294
|
|
|
214
295
|
---
|
|
215
296
|
|
|
@@ -246,5 +327,4 @@ that as `WARN (PARTIAL)` instead of accepting the citation.
|
|
|
246
327
|
|
|
247
328
|
## Related
|
|
248
329
|
|
|
249
|
-
- [
|
|
250
|
-
- [decide-skill](https://github.com/Moonweave-Systems/decide-skill) - decision automation for non-expert domains
|
|
330
|
+
- [decision-kernel](https://github.com/Moonweave-Systems/decision-kernel) - evidence-gated decisions and drift/done checks for coding agents
|
|
@@ -34,6 +34,10 @@ Works with **Claude Code, Cursor, Codex**, and any agent that supports the
|
|
|
34
34
|
|
|
35
35
|
After installation, use it like a normal agent skill. You do not start a server and you do not configure MCP for this workflow. No MCP server is required for this workflow.
|
|
36
36
|
|
|
37
|
+
The skill includes its own copy of the CLI engine, and the agent runs it from the skill folder, so nothing else needs to be installed. Python 3.10 or newer must be available as `python3`.
|
|
38
|
+
|
|
39
|
+
For explicit agent tool-calling rules, see [AGENT_USAGE.md](https://github.com/Moonweave-Research/ref-verify/blob/main/AGENT_USAGE.md).
|
|
40
|
+
|
|
37
41
|
---
|
|
38
42
|
|
|
39
43
|
## Use it
|
|
@@ -64,20 +68,28 @@ checks that are currently safe to automate directly:
|
|
|
64
68
|
|
|
65
69
|
- CrossRef metadata check: `ref-verify verify-doi`
|
|
66
70
|
- DOI-bound abstract claim check: `ref-verify check-claim`
|
|
71
|
+
- Batch DOI-bound claim checks: `ref-verify check-file`
|
|
67
72
|
- literal text claims
|
|
68
73
|
- subject-matched percentage claims such as efficiency, response rate, or actuation strain
|
|
69
74
|
- simple unit/count claims such as cycles, patients, voltage, temperature, and concentration
|
|
70
|
-
- CrossRef first, then DOI-bound Semantic Scholar and PubMed fallback when CrossRef has no abstract
|
|
75
|
+
- CrossRef first, then DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback when CrossRef has no abstract
|
|
71
76
|
- JSON output for agent-readable routing
|
|
72
77
|
- Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
|
|
73
78
|
|
|
74
|
-
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol.
|
|
79
|
+
Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ratio, and confidence intervals still use the manual skill protocol. DOI landing-page checks still use the skill protocol. The CLI rejects a DOI that CrossRef records as retracted (via its retraction notice); retraction banners CrossRef does not know about, Unpaywall, arXiv, and two-source existence checks remain in the `SKILL.md` protocol.
|
|
75
80
|
|
|
76
81
|
The CLI has zero third-party Python runtime dependencies, but it is not an
|
|
77
82
|
offline verifier. Functional checks require outbound HTTPS access to public
|
|
78
|
-
academic APIs such as CrossRef, Semantic Scholar, and PubMed.
|
|
83
|
+
academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
|
|
79
84
|
|
|
80
|
-
|
|
85
|
+
To run the CLI yourself, install it from PyPI:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
uvx ref-verify --help # run without installing (uv)
|
|
89
|
+
pipx install ref-verify # or install the `ref-verify` command
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Or install it from a local checkout:
|
|
81
93
|
|
|
82
94
|
```bash
|
|
83
95
|
git clone https://github.com/Moonweave-Research/ref-verify.git
|
|
@@ -116,7 +128,7 @@ ref-verify check-claim 10.1126/science.287.5454.836 \
|
|
|
116
128
|
--json
|
|
117
129
|
```
|
|
118
130
|
|
|
119
|
-
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound Semantic Scholar and PubMed fallback sources. Use `--source crossref`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
131
|
+
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback sources. Use `--source crossref`, `--source openalex`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
120
132
|
|
|
121
133
|
Source-checkout equivalents:
|
|
122
134
|
|
|
@@ -158,6 +170,50 @@ normal CI does not fail because an upstream API is temporarily unavailable.
|
|
|
158
170
|
|
|
159
171
|
---
|
|
160
172
|
|
|
173
|
+
## Scope — optional CLI versus manual audit
|
|
174
|
+
|
|
175
|
+
`ref-verify` is a conservative guard, not an oracle. It errs toward flagging: an
|
|
176
|
+
`ACCEPT` is high-confidence, and **anything else means "not auto-verifiable —
|
|
177
|
+
check it yourself", not "the citation is wrong."**
|
|
178
|
+
|
|
179
|
+
**The optional CLI verifies**
|
|
180
|
+
|
|
181
|
+
- DOI metadata: title, first-author surname, and year against CrossRef.
|
|
182
|
+
- Whether a DOI-bound **abstract** explicitly supports a specific numeric or
|
|
183
|
+
literal claim, quoted verbatim. If no abstract is reachable, it returns
|
|
184
|
+
`UNVERIFIABLE` rather than guessing.
|
|
185
|
+
|
|
186
|
+
**The optional CLI does not verify** (out of scope by design, not bugs)
|
|
187
|
+
|
|
188
|
+
- **Full-text, figure, table, or supplementary values** — abstract-only. A number
|
|
189
|
+
that appears only in the body stays `UNVERIFIABLE`.
|
|
190
|
+
- **Relational or qualitative claims** — proportionalities, mechanisms,
|
|
191
|
+
"broader/stronger than". Only value+unit and literal claims are checked.
|
|
192
|
+
- **Papers whose publisher withholds the abstract** — some titles expose no
|
|
193
|
+
abstract to CrossRef or OpenAlex. No abstract → `UNVERIFIABLE`, which reflects
|
|
194
|
+
reachability, not the claim.
|
|
195
|
+
- **Statistical metrics** (p-value, AUC/AUROC, F1, hazard/odds ratio, confidence
|
|
196
|
+
intervals) — handled by the manual skill protocol, not the CLI.
|
|
197
|
+
- **Paper quality, novelty, field consensus**, or whether the *full* paper
|
|
198
|
+
supports a broader statement.
|
|
199
|
+
|
|
200
|
+
The agent skill's manual Full Audit protocol goes beyond the optional CLI for
|
|
201
|
+
mechanism, implementation, and procedural claims. It requires a fetched
|
|
202
|
+
full-text passage at that source depth; when full text is unavailable, it
|
|
203
|
+
returns `WARN (ABSTRACT-ONLY)` instead of upgrading an abstract topic match to
|
|
204
|
+
`ACCEPT`.
|
|
205
|
+
|
|
206
|
+
**Reading a CLI verdict**
|
|
207
|
+
|
|
208
|
+
| Verdict | Meaning |
|
|
209
|
+
|---|---|
|
|
210
|
+
| `ACCEPT` | The fetched abstract explicitly supports the claim. High-confidence pass. |
|
|
211
|
+
| `WARN` / `PARTIAL` | An abstract was read but does not explicitly support the exact claim. Check the source. |
|
|
212
|
+
| `UNVERIFIABLE` | No abstract was reachable to check against. Not a judgment on the claim. |
|
|
213
|
+
| `REJECT` | DOI is dead, resolves to a different paper, contradicted, or retracted. |
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
161
217
|
## Modes
|
|
162
218
|
|
|
163
219
|
**Quick Screen** is for DOIs you already have. It uses CrossRef to compare the
|
|
@@ -171,9 +227,10 @@ ref-verify verify-doi <doi> --title "<title>" --first-author <last-name> --year
|
|
|
171
227
|
exit code, so weak or mismatched metadata cannot silently pass automation gates.
|
|
172
228
|
|
|
173
229
|
**Full Audit** is for literature search and final pre-submission review. The
|
|
174
|
-
skill fetches abstracts through CrossRef, Semantic Scholar, Unpaywall,
|
|
175
|
-
and PubMed where needed
|
|
176
|
-
claim
|
|
230
|
+
skill fetches abstracts through CrossRef, OpenAlex, Semantic Scholar, Unpaywall,
|
|
231
|
+
arXiv, and PubMed where needed. For a topline claim, it checks the abstract; for
|
|
232
|
+
a mechanism, implementation, or procedural claim, it continues to a fetched
|
|
233
|
+
full-text passage before assigning support.
|
|
177
234
|
|
|
178
235
|
For a single DOI-backed claim, the CLI can run the abstract check:
|
|
179
236
|
|
|
@@ -186,19 +243,43 @@ ref-verify check-claim <doi> --claim "<specific claim>" --json
|
|
|
186
243
|
`abstract_source`, `source_attempts`, and `error_code` so agents can distinguish
|
|
187
244
|
missing abstracts, source failures, DOI mismatches, and ambiguous evidence.
|
|
188
245
|
|
|
246
|
+
Use `check-file` when a draft, literature note, or AI-agent output has many
|
|
247
|
+
DOI/claim pairs.
|
|
248
|
+
|
|
249
|
+
JSONL:
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
ref-verify check-file claims.jsonl
|
|
253
|
+
ref-verify check-file claims.jsonl --json
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
CSV:
|
|
257
|
+
|
|
258
|
+
```bash
|
|
259
|
+
ref-verify check-file claims.csv
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
Each row must include `doi` and `claim`. Optional fields are `id`, `source`,
|
|
263
|
+
and `note`. Batch mode reuses the same conservative `check-claim` engine:
|
|
264
|
+
`ACCEPT` means the abstract explicitly supports the numeric claim. `WARN`,
|
|
265
|
+
`PARTIAL`, `REJECT`, or `UNVERIFIABLE` means the claim should not be treated as
|
|
266
|
+
verified.
|
|
267
|
+
|
|
189
268
|
Current `check-claim` error codes:
|
|
190
269
|
|
|
191
270
|
- `CLAIM_SUPPORTED`: explicit abstract support found.
|
|
192
271
|
- `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
|
|
193
272
|
- `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
|
|
194
273
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
195
|
-
- `DOI_NOT_FOUND`: selected source did not find a DOI-bound record.
|
|
274
|
+
- `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
|
|
275
|
+
- `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
|
|
196
276
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
197
|
-
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_UNSUPPORTED`: source lookup failed or could not be used.
|
|
277
|
+
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
198
278
|
|
|
199
279
|
> Core rule: every content statement about a paper must come from a live-fetched
|
|
200
|
-
>
|
|
201
|
-
>
|
|
280
|
+
> source at the depth the claim requires — abstract for topline claims, full
|
|
281
|
+
> text for mechanism, implementation, or procedural claims. If the required
|
|
282
|
+
> source is inaccessible, say so. Do not fill the gap from memory.
|
|
202
283
|
|
|
203
284
|
---
|
|
204
285
|
|
|
@@ -235,5 +316,4 @@ that as `WARN (PARTIAL)` instead of accepting the citation.
|
|
|
235
316
|
|
|
236
317
|
## Related
|
|
237
318
|
|
|
238
|
-
- [
|
|
239
|
-
- [decide-skill](https://github.com/Moonweave-Systems/decide-skill) - decision automation for non-expert domains
|
|
319
|
+
- [decision-kernel](https://github.com/Moonweave-Systems/decision-kernel) - evidence-gated decisions and drift/done checks for coding agents
|
|
@@ -202,6 +202,8 @@ def _final_error_code(
|
|
|
202
202
|
return "SOURCE_API_ERROR"
|
|
203
203
|
if "TIMEOUT" in statuses:
|
|
204
204
|
return "SOURCE_TIMEOUT"
|
|
205
|
+
if "RATE_LIMITED" in statuses:
|
|
206
|
+
return "SOURCE_RATE_LIMITED"
|
|
205
207
|
if "UNSUPPORTED" in statuses and "NO_ABSTRACT" not in statuses:
|
|
206
208
|
return "SOURCE_UNSUPPORTED"
|
|
207
209
|
if statuses and all(status == "NOT_FOUND" for status in statuses):
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import asdict, dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Literal
|
|
8
|
+
|
|
9
|
+
BatchFormat = Literal["jsonl", "csv"]
|
|
10
|
+
_VALID_SOURCES = {"auto", "crossref", "openalex", "semantic-scholar", "pubmed"}
|
|
11
|
+
_FAILED_ERROR_CODES = {
|
|
12
|
+
"ROW_CHECK_ERROR",
|
|
13
|
+
"SOURCE_API_ERROR",
|
|
14
|
+
"SOURCE_TIMEOUT",
|
|
15
|
+
"SOURCE_RATE_LIMITED",
|
|
16
|
+
"SOURCE_UNSUPPORTED",
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class BatchInputError(ValueError):
|
|
21
|
+
pass
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class ClaimInputRow:
|
|
26
|
+
row_number: int
|
|
27
|
+
id: str | None
|
|
28
|
+
doi: str
|
|
29
|
+
claim: str
|
|
30
|
+
source: str = "auto"
|
|
31
|
+
note: str | None = None
|
|
32
|
+
|
|
33
|
+
def to_dict(self) -> dict[str, Any]:
|
|
34
|
+
return asdict(self)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class BatchRowResult:
|
|
39
|
+
row: ClaimInputRow
|
|
40
|
+
payload: dict[str, Any]
|
|
41
|
+
|
|
42
|
+
def to_dict(self) -> dict[str, Any]:
|
|
43
|
+
result = {
|
|
44
|
+
"row_number": self.row.row_number,
|
|
45
|
+
"id": self.row.id,
|
|
46
|
+
"doi": self.row.doi,
|
|
47
|
+
"claim": self.row.claim,
|
|
48
|
+
}
|
|
49
|
+
if self.row.note is not None:
|
|
50
|
+
result["note"] = self.row.note
|
|
51
|
+
result.update(self.payload)
|
|
52
|
+
return result
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(frozen=True)
|
|
56
|
+
class BatchSummary:
|
|
57
|
+
total: int
|
|
58
|
+
accept: int
|
|
59
|
+
warn: int
|
|
60
|
+
reject: int
|
|
61
|
+
partial: int
|
|
62
|
+
unverifiable: int
|
|
63
|
+
failed: int
|
|
64
|
+
|
|
65
|
+
def to_dict(self) -> dict[str, int]:
|
|
66
|
+
return asdict(self)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def detect_format(path: Path, explicit_format: str | None) -> BatchFormat:
|
|
70
|
+
if explicit_format in ("jsonl", "csv"):
|
|
71
|
+
return explicit_format
|
|
72
|
+
if explicit_format is not None:
|
|
73
|
+
raise BatchInputError(f"Unsupported input format: {explicit_format}")
|
|
74
|
+
suffix = path.suffix.lower()
|
|
75
|
+
if suffix == ".jsonl":
|
|
76
|
+
return "jsonl"
|
|
77
|
+
if suffix == ".csv":
|
|
78
|
+
return "csv"
|
|
79
|
+
raise BatchInputError("Unsupported input format; use .jsonl, .csv, or --format")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def parse_claim_file(path: Path, explicit_format: str | None) -> list[ClaimInputRow]:
|
|
83
|
+
batch_format = detect_format(path, explicit_format)
|
|
84
|
+
try:
|
|
85
|
+
if batch_format == "jsonl":
|
|
86
|
+
rows = _parse_jsonl(path)
|
|
87
|
+
else:
|
|
88
|
+
rows = _parse_csv(path)
|
|
89
|
+
except OSError as exc:
|
|
90
|
+
raise BatchInputError(f"Could not read input file: {exc}") from exc
|
|
91
|
+
if not rows:
|
|
92
|
+
raise BatchInputError("Input file does not contain any claim rows")
|
|
93
|
+
return rows
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def summarize_results(results: list[BatchRowResult]) -> BatchSummary:
|
|
97
|
+
accept = warn = reject = partial = unverifiable = failed = 0
|
|
98
|
+
for result in results:
|
|
99
|
+
verdict = str(result.payload.get("verdict", ""))
|
|
100
|
+
status = str(result.payload.get("status", ""))
|
|
101
|
+
if verdict == "ACCEPT":
|
|
102
|
+
accept += 1
|
|
103
|
+
if verdict == "WARN":
|
|
104
|
+
warn += 1
|
|
105
|
+
if verdict == "REJECT":
|
|
106
|
+
reject += 1
|
|
107
|
+
if status == "PARTIAL":
|
|
108
|
+
partial += 1
|
|
109
|
+
if status == "UNVERIFIABLE":
|
|
110
|
+
unverifiable += 1
|
|
111
|
+
if verdict == "ERROR" or str(result.payload.get("error_code", "")) in _FAILED_ERROR_CODES:
|
|
112
|
+
failed += 1
|
|
113
|
+
return BatchSummary(
|
|
114
|
+
total=len(results),
|
|
115
|
+
accept=accept,
|
|
116
|
+
warn=warn,
|
|
117
|
+
reject=reject,
|
|
118
|
+
partial=partial,
|
|
119
|
+
unverifiable=unverifiable,
|
|
120
|
+
failed=failed,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def batch_payload(results: list[BatchRowResult]) -> dict[str, Any]:
|
|
125
|
+
return {
|
|
126
|
+
"summary": summarize_results(results).to_dict(),
|
|
127
|
+
"results": [result.to_dict() for result in results],
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def render_batch_text(results: list[BatchRowResult]) -> str:
|
|
132
|
+
summary = summarize_results(results)
|
|
133
|
+
lines = [
|
|
134
|
+
(
|
|
135
|
+
f"Summary: total={summary.total} accept={summary.accept} "
|
|
136
|
+
f"warn={summary.warn} reject={summary.reject} "
|
|
137
|
+
f"partial={summary.partial} unverifiable={summary.unverifiable} "
|
|
138
|
+
f"failed={summary.failed}"
|
|
139
|
+
)
|
|
140
|
+
]
|
|
141
|
+
for result in results:
|
|
142
|
+
payload = result.payload
|
|
143
|
+
label = str(payload.get("verdict", "WARN"))
|
|
144
|
+
row_id = result.row.id or f"row-{result.row.row_number}"
|
|
145
|
+
lines.extend(
|
|
146
|
+
[
|
|
147
|
+
"",
|
|
148
|
+
f"{label} {row_id} {result.row.doi}",
|
|
149
|
+
f"Claim: {result.row.claim}",
|
|
150
|
+
f"Reason: {payload.get('reason', '')}",
|
|
151
|
+
]
|
|
152
|
+
)
|
|
153
|
+
evidence = payload.get("evidence")
|
|
154
|
+
if evidence:
|
|
155
|
+
lines.append(f"Evidence: {evidence}")
|
|
156
|
+
error_code = payload.get("error_code")
|
|
157
|
+
if error_code:
|
|
158
|
+
lines.append(f"Error code: {error_code}")
|
|
159
|
+
return "\n".join(lines)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _parse_jsonl(path: Path) -> list[ClaimInputRow]:
|
|
163
|
+
rows: list[ClaimInputRow] = []
|
|
164
|
+
with path.open("r", encoding="utf-8") as handle:
|
|
165
|
+
for line_number, line in enumerate(handle, start=1):
|
|
166
|
+
stripped = line.strip()
|
|
167
|
+
if not stripped:
|
|
168
|
+
continue
|
|
169
|
+
try:
|
|
170
|
+
raw = json.loads(stripped)
|
|
171
|
+
except json.JSONDecodeError as exc:
|
|
172
|
+
raise BatchInputError(f"Invalid JSON on line {line_number}: {exc.msg}") from exc
|
|
173
|
+
if not isinstance(raw, dict):
|
|
174
|
+
raise BatchInputError(f"Invalid row on line {line_number}: expected object")
|
|
175
|
+
rows.append(_row_from_mapping(raw, line_number=line_number, row_label="line"))
|
|
176
|
+
return rows
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _parse_csv(path: Path) -> list[ClaimInputRow]:
|
|
180
|
+
rows: list[ClaimInputRow] = []
|
|
181
|
+
with path.open("r", encoding="utf-8-sig", newline="") as handle:
|
|
182
|
+
reader = csv.DictReader(handle, skipinitialspace=True)
|
|
183
|
+
if reader.fieldnames is None:
|
|
184
|
+
raise BatchInputError("CSV input is missing a header row")
|
|
185
|
+
fieldnames = {_normalize_csv_fieldname(field) for field in reader.fieldnames if field}
|
|
186
|
+
missing = {"doi", "claim"} - fieldnames
|
|
187
|
+
if missing:
|
|
188
|
+
missing_fields = ", ".join(sorted(missing))
|
|
189
|
+
raise BatchInputError(f"CSV header must include doi and claim fields; missing: {missing_fields}")
|
|
190
|
+
for row_number, raw in enumerate(reader, start=2):
|
|
191
|
+
rows.append(
|
|
192
|
+
_row_from_mapping(
|
|
193
|
+
_normalize_csv_row(raw),
|
|
194
|
+
line_number=row_number,
|
|
195
|
+
row_label="line",
|
|
196
|
+
)
|
|
197
|
+
)
|
|
198
|
+
return rows
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _row_from_mapping(raw: dict[str, Any], *, line_number: int, row_label: str) -> ClaimInputRow:
|
|
202
|
+
doi = _required_string(raw, "doi", line_number, row_label)
|
|
203
|
+
claim = _required_string(raw, "claim", line_number, row_label)
|
|
204
|
+
source = _optional_string(raw, "source") or "auto"
|
|
205
|
+
if source not in _VALID_SOURCES:
|
|
206
|
+
raise BatchInputError(f"Invalid field on {row_label} {line_number}: source={source}")
|
|
207
|
+
return ClaimInputRow(
|
|
208
|
+
row_number=line_number,
|
|
209
|
+
id=_optional_string(raw, "id"),
|
|
210
|
+
doi=doi,
|
|
211
|
+
claim=claim,
|
|
212
|
+
source=source,
|
|
213
|
+
note=_optional_string(raw, "note"),
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _required_string(raw: dict[str, Any], field: str, line_number: int, row_label: str) -> str:
|
|
218
|
+
value = raw.get(field)
|
|
219
|
+
if not isinstance(value, str) or not value.strip():
|
|
220
|
+
raise BatchInputError(f"Missing required field on {row_label} {line_number}: {field}")
|
|
221
|
+
return value.strip()
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _optional_string(raw: dict[str, Any], field: str) -> str | None:
|
|
225
|
+
value = raw.get(field)
|
|
226
|
+
if value is None:
|
|
227
|
+
return None
|
|
228
|
+
if not isinstance(value, str):
|
|
229
|
+
return str(value)
|
|
230
|
+
stripped = value.strip()
|
|
231
|
+
return stripped or None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _normalize_csv_fieldname(field: str) -> str:
|
|
235
|
+
return field.lstrip("\ufeff").strip()
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _normalize_csv_row(raw: dict[str, Any]) -> dict[str, Any]:
|
|
239
|
+
return {_normalize_csv_fieldname(key): value for key, value in raw.items() if key is not None}
|
|
@@ -209,13 +209,29 @@ _UNSUPPORTED_CLAIM_FRAME_PATTERNS = (
|
|
|
209
209
|
r"\bsaid to\b",
|
|
210
210
|
r"\b(?:expect|expects|expected|expecting) to\b",
|
|
211
211
|
r"\b(?:project|projects|projected|projecting) to\b",
|
|
212
|
-
r"\b(?:estimate|estimates|estimated|estimating) to\b",
|
|
212
|
+
r"\b(?:estimate|estimates|estimated|estimating) to (?!be\b|equal\b|at\b|about\b)",
|
|
213
213
|
r"\bachievable\b",
|
|
214
214
|
r"\bpossible\b",
|
|
215
215
|
)
|
|
216
216
|
|
|
217
217
|
|
|
218
|
+
def retracted_claim_result(record: PaperRecord, claim: str) -> ClaimSupportResult:
|
|
219
|
+
return ClaimSupportResult(
|
|
220
|
+
status="RETRACTED",
|
|
221
|
+
verdict="REJECT",
|
|
222
|
+
reason=(
|
|
223
|
+
"CrossRef records this paper as retracted "
|
|
224
|
+
f"(notice DOI {record.retraction_doi}); its abstract cannot support the claim."
|
|
225
|
+
),
|
|
226
|
+
evidence="",
|
|
227
|
+
paper=record,
|
|
228
|
+
claim=claim,
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
|
|
218
232
|
def check_claim_support(record: PaperRecord, claim: str) -> ClaimSupportResult:
|
|
233
|
+
if record.retraction_doi:
|
|
234
|
+
return retracted_claim_result(record, claim)
|
|
219
235
|
if not record.abstract:
|
|
220
236
|
return ClaimSupportResult(
|
|
221
237
|
status="UNVERIFIABLE",
|
|
@@ -338,6 +354,8 @@ def _claim_percentage_comparator(claim: str) -> str:
|
|
|
338
354
|
return "gte"
|
|
339
355
|
if re.search(r"\b(at most|no more than)\b", normalized):
|
|
340
356
|
return "lte"
|
|
357
|
+
if re.search(r"\bup to\b", normalized):
|
|
358
|
+
return "up_to"
|
|
341
359
|
if re.search(r"\b(below|under|less than)\b", normalized):
|
|
342
360
|
return "lt"
|
|
343
361
|
if re.search(
|
|
@@ -414,6 +432,8 @@ def _evidence_entails_claim(
|
|
|
414
432
|
claim_comparator: str,
|
|
415
433
|
) -> bool:
|
|
416
434
|
if evidence_comparator == "up_to":
|
|
435
|
+
if claim_comparator == "up_to":
|
|
436
|
+
return value == threshold
|
|
417
437
|
if claim_comparator == "exact":
|
|
418
438
|
return False
|
|
419
439
|
return _compare_percentage(value, threshold, claim_comparator)
|