ref-verify 1.1.2__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ref_verify-1.1.2/src/ref_verify.egg-info → ref_verify-1.2.0}/PKG-INFO +70 -7
- {ref_verify-1.1.2 → ref_verify-1.2.0}/README.md +69 -6
- {ref_verify-1.1.2 → ref_verify-1.2.0}/pyproject.toml +1 -1
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/__init__.py +1 -1
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/abstract_lookup.py +2 -0
- ref_verify-1.2.0/src/ref_verify/batch.py +239 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/claim_check.py +5 -1
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/cli.py +77 -11
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/numeric_claim.py +133 -13
- ref_verify-1.2.0/src/ref_verify/openalex.py +123 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/semantic_scholar.py +41 -14
- {ref_verify-1.1.2 → ref_verify-1.2.0/src/ref_verify.egg-info}/PKG-INFO +70 -7
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify.egg-info/SOURCES.txt +3 -0
- ref_verify-1.2.0/tests/test_abstract_sources.py +234 -0
- ref_verify-1.2.0/tests/test_batch.py +238 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/tests/test_claim_check.py +36 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/tests/test_cli.py +423 -1
- ref_verify-1.2.0/tests/test_numeric_claim.py +252 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/tests/test_package_smoke.py +6 -2
- {ref_verify-1.1.2 → ref_verify-1.2.0}/tests/test_skill_docs.py +39 -7
- ref_verify-1.1.2/tests/test_abstract_sources.py +0 -105
- ref_verify-1.1.2/tests/test_numeric_claim.py +0 -88
- {ref_verify-1.1.2 → ref_verify-1.2.0}/LICENSE +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/setup.cfg +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/crossref.py +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/doi_check.py +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/models.py +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify/pubmed.py +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify.egg-info/dependency_links.txt +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify.egg-info/entry_points.txt +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/src/ref_verify.egg-info/top_level.txt +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/tests/test_crossref.py +0 -0
- {ref_verify-1.1.2 → ref_verify-1.2.0}/tests/test_doi_check.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ref-verify
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Executable DOI and claim verification helpers for academic citations
|
|
5
5
|
Author: Moonweave Research
|
|
6
6
|
License-Expression: MIT
|
|
@@ -45,6 +45,8 @@ Works with **Claude Code, Cursor, Codex**, and any agent that supports the
|
|
|
45
45
|
|
|
46
46
|
After installation, use it like a normal agent skill. You do not start a server and you do not configure MCP for this workflow. No MCP server is required for this workflow.
|
|
47
47
|
|
|
48
|
+
For explicit agent tool-calling rules, see [AGENT_USAGE.md](https://github.com/Moonweave-Research/ref-verify/blob/main/AGENT_USAGE.md).
|
|
49
|
+
|
|
48
50
|
---
|
|
49
51
|
|
|
50
52
|
## Use it
|
|
@@ -75,10 +77,11 @@ checks that are currently safe to automate directly:
|
|
|
75
77
|
|
|
76
78
|
- CrossRef metadata check: `ref-verify verify-doi`
|
|
77
79
|
- DOI-bound abstract claim check: `ref-verify check-claim`
|
|
80
|
+
- Batch DOI-bound claim checks: `ref-verify check-file`
|
|
78
81
|
- literal text claims
|
|
79
82
|
- subject-matched percentage claims such as efficiency, response rate, or actuation strain
|
|
80
83
|
- simple unit/count claims such as cycles, patients, voltage, temperature, and concentration
|
|
81
|
-
- CrossRef first, then DOI-bound Semantic Scholar and PubMed fallback when CrossRef has no abstract
|
|
84
|
+
- CrossRef first, then DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback when CrossRef has no abstract
|
|
82
85
|
- JSON output for agent-readable routing
|
|
83
86
|
- Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
|
|
84
87
|
|
|
@@ -86,7 +89,7 @@ Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ra
|
|
|
86
89
|
|
|
87
90
|
The CLI has zero third-party Python runtime dependencies, but it is not an
|
|
88
91
|
offline verifier. Functional checks require outbound HTTPS access to public
|
|
89
|
-
academic APIs such as CrossRef, Semantic Scholar, and PubMed.
|
|
92
|
+
academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
|
|
90
93
|
|
|
91
94
|
Install the CLI from a local checkout:
|
|
92
95
|
|
|
@@ -127,7 +130,7 @@ ref-verify check-claim 10.1126/science.287.5454.836 \
|
|
|
127
130
|
--json
|
|
128
131
|
```
|
|
129
132
|
|
|
130
|
-
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound Semantic Scholar and PubMed fallback sources. Use `--source crossref`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
133
|
+
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback sources. Use `--source crossref`, `--source openalex`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
131
134
|
|
|
132
135
|
Source-checkout equivalents:
|
|
133
136
|
|
|
@@ -169,6 +172,44 @@ normal CI does not fail because an upstream API is temporarily unavailable.
|
|
|
169
172
|
|
|
170
173
|
---
|
|
171
174
|
|
|
175
|
+
## Scope — what it does and does not verify
|
|
176
|
+
|
|
177
|
+
`ref-verify` is a conservative guard, not an oracle. It errs toward flagging: an
|
|
178
|
+
`ACCEPT` is high-confidence, and **anything else means "not auto-verifiable —
|
|
179
|
+
check it yourself", not "the citation is wrong."**
|
|
180
|
+
|
|
181
|
+
**It verifies**
|
|
182
|
+
|
|
183
|
+
- DOI metadata: title, first-author surname, and year against CrossRef.
|
|
184
|
+
- Whether a DOI-bound **abstract** explicitly supports a specific numeric or
|
|
185
|
+
literal claim, quoted verbatim. If no abstract is reachable, it returns
|
|
186
|
+
`UNVERIFIABLE` rather than guessing.
|
|
187
|
+
|
|
188
|
+
**It does not verify** (out of scope by design, not bugs)
|
|
189
|
+
|
|
190
|
+
- **Full-text, figure, table, or supplementary values** — abstract-only. A number
|
|
191
|
+
that appears only in the body stays `UNVERIFIABLE`.
|
|
192
|
+
- **Relational or qualitative claims** — proportionalities, mechanisms,
|
|
193
|
+
"broader/stronger than". Only value+unit and literal claims are checked.
|
|
194
|
+
- **Papers whose publisher withholds the abstract** — some titles expose no
|
|
195
|
+
abstract to CrossRef or OpenAlex. No abstract → `UNVERIFIABLE`, which reflects
|
|
196
|
+
reachability, not the claim.
|
|
197
|
+
- **Statistical metrics** (p-value, AUC/AUROC, F1, hazard/odds ratio, confidence
|
|
198
|
+
intervals) — handled by the manual skill protocol, not the CLI.
|
|
199
|
+
- **Paper quality, novelty, field consensus**, or whether the *full* paper
|
|
200
|
+
supports a broader statement.
|
|
201
|
+
|
|
202
|
+
**Reading a verdict**
|
|
203
|
+
|
|
204
|
+
| Verdict | Meaning |
|
|
205
|
+
|---|---|
|
|
206
|
+
| `ACCEPT` | The fetched abstract explicitly supports the claim. High-confidence pass. |
|
|
207
|
+
| `WARN` / `PARTIAL` | An abstract was read but does not explicitly support the exact claim. Check the source. |
|
|
208
|
+
| `UNVERIFIABLE` | No abstract was reachable to check against. Not a judgment on the claim. |
|
|
209
|
+
| `REJECT` | DOI is dead, resolves to a different paper, contradicted, or retracted. |
|
|
210
|
+
|
|
211
|
+
---
|
|
212
|
+
|
|
172
213
|
## Modes
|
|
173
214
|
|
|
174
215
|
**Quick Screen** is for DOIs you already have. It uses CrossRef to compare the
|
|
@@ -182,8 +223,8 @@ ref-verify verify-doi <doi> --title "<title>" --first-author <last-name> --year
|
|
|
182
223
|
exit code, so weak or mismatched metadata cannot silently pass automation gates.
|
|
183
224
|
|
|
184
225
|
**Full Audit** is for literature search and final pre-submission review. The
|
|
185
|
-
skill fetches abstracts through CrossRef, Semantic Scholar, Unpaywall,
|
|
186
|
-
and PubMed where needed, then checks whether the paper supports the specific
|
|
226
|
+
skill fetches abstracts through CrossRef, OpenAlex, Semantic Scholar, Unpaywall,
|
|
227
|
+
arXiv, and PubMed where needed, then checks whether the paper supports the specific
|
|
187
228
|
claim being cited.
|
|
188
229
|
|
|
189
230
|
For a single DOI-backed claim, the CLI can run the abstract check:
|
|
@@ -197,6 +238,28 @@ ref-verify check-claim <doi> --claim "<specific claim>" --json
|
|
|
197
238
|
`abstract_source`, `source_attempts`, and `error_code` so agents can distinguish
|
|
198
239
|
missing abstracts, source failures, DOI mismatches, and ambiguous evidence.
|
|
199
240
|
|
|
241
|
+
Use `check-file` when a draft, literature note, or AI-agent output has many
|
|
242
|
+
DOI/claim pairs.
|
|
243
|
+
|
|
244
|
+
JSONL:
|
|
245
|
+
|
|
246
|
+
```bash
|
|
247
|
+
ref-verify check-file claims.jsonl
|
|
248
|
+
ref-verify check-file claims.jsonl --json
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
CSV:
|
|
252
|
+
|
|
253
|
+
```bash
|
|
254
|
+
ref-verify check-file claims.csv
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
Each row must include `doi` and `claim`. Optional fields are `id`, `source`,
|
|
258
|
+
and `note`. Batch mode reuses the same conservative `check-claim` engine:
|
|
259
|
+
`ACCEPT` means the abstract explicitly supports the numeric claim. `WARN`,
|
|
260
|
+
`PARTIAL`, `REJECT`, or `UNVERIFIABLE` means the claim should not be treated as
|
|
261
|
+
verified.
|
|
262
|
+
|
|
200
263
|
Current `check-claim` error codes:
|
|
201
264
|
|
|
202
265
|
- `CLAIM_SUPPORTED`: explicit abstract support found.
|
|
@@ -205,7 +268,7 @@ Current `check-claim` error codes:
|
|
|
205
268
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
206
269
|
- `DOI_NOT_FOUND`: selected source did not find a DOI-bound record.
|
|
207
270
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
208
|
-
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_UNSUPPORTED`: source lookup failed or could not be used.
|
|
271
|
+
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
209
272
|
|
|
210
273
|
> Core rule: every content statement about a paper must come from a live-fetched
|
|
211
274
|
> abstract. If the abstract is inaccessible after fallback checks, say
|
|
@@ -34,6 +34,8 @@ Works with **Claude Code, Cursor, Codex**, and any agent that supports the
|
|
|
34
34
|
|
|
35
35
|
After installation, use it like a normal agent skill. You do not start a server and you do not configure MCP for this workflow. No MCP server is required for this workflow.
|
|
36
36
|
|
|
37
|
+
For explicit agent tool-calling rules, see [AGENT_USAGE.md](https://github.com/Moonweave-Research/ref-verify/blob/main/AGENT_USAGE.md).
|
|
38
|
+
|
|
37
39
|
---
|
|
38
40
|
|
|
39
41
|
## Use it
|
|
@@ -64,10 +66,11 @@ checks that are currently safe to automate directly:
|
|
|
64
66
|
|
|
65
67
|
- CrossRef metadata check: `ref-verify verify-doi`
|
|
66
68
|
- DOI-bound abstract claim check: `ref-verify check-claim`
|
|
69
|
+
- Batch DOI-bound claim checks: `ref-verify check-file`
|
|
67
70
|
- literal text claims
|
|
68
71
|
- subject-matched percentage claims such as efficiency, response rate, or actuation strain
|
|
69
72
|
- simple unit/count claims such as cycles, patients, voltage, temperature, and concentration
|
|
70
|
-
- CrossRef first, then DOI-bound Semantic Scholar and PubMed fallback when CrossRef has no abstract
|
|
73
|
+
- CrossRef first, then DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback when CrossRef has no abstract
|
|
71
74
|
- JSON output for agent-readable routing
|
|
72
75
|
- Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
|
|
73
76
|
|
|
@@ -75,7 +78,7 @@ Statistical metrics such as p-values, AUC/AUROC, F1 score, hazard ratio, odds ra
|
|
|
75
78
|
|
|
76
79
|
The CLI has zero third-party Python runtime dependencies, but it is not an
|
|
77
80
|
offline verifier. Functional checks require outbound HTTPS access to public
|
|
78
|
-
academic APIs such as CrossRef, Semantic Scholar, and PubMed.
|
|
81
|
+
academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
|
|
79
82
|
|
|
80
83
|
Install the CLI from a local checkout:
|
|
81
84
|
|
|
@@ -116,7 +119,7 @@ ref-verify check-claim 10.1126/science.287.5454.836 \
|
|
|
116
119
|
--json
|
|
117
120
|
```
|
|
118
121
|
|
|
119
|
-
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound Semantic Scholar and PubMed fallback sources. Use `--source crossref`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
122
|
+
By default, `check-claim` uses CrossRef first. If CrossRef has no abstract, it tries DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback sources. Use `--source crossref`, `--source openalex`, `--source semantic-scholar`, or `--source pubmed` for source-specific debugging; explicit non-CrossRef source selection bypasses CrossRef.
|
|
120
123
|
|
|
121
124
|
Source-checkout equivalents:
|
|
122
125
|
|
|
@@ -158,6 +161,44 @@ normal CI does not fail because an upstream API is temporarily unavailable.
|
|
|
158
161
|
|
|
159
162
|
---
|
|
160
163
|
|
|
164
|
+
## Scope — what it does and does not verify
|
|
165
|
+
|
|
166
|
+
`ref-verify` is a conservative guard, not an oracle. It errs toward flagging: an
|
|
167
|
+
`ACCEPT` is high-confidence, and **anything else means "not auto-verifiable —
|
|
168
|
+
check it yourself", not "the citation is wrong."**
|
|
169
|
+
|
|
170
|
+
**It verifies**
|
|
171
|
+
|
|
172
|
+
- DOI metadata: title, first-author surname, and year against CrossRef.
|
|
173
|
+
- Whether a DOI-bound **abstract** explicitly supports a specific numeric or
|
|
174
|
+
literal claim, quoted verbatim. If no abstract is reachable, it returns
|
|
175
|
+
`UNVERIFIABLE` rather than guessing.
|
|
176
|
+
|
|
177
|
+
**It does not verify** (out of scope by design, not bugs)
|
|
178
|
+
|
|
179
|
+
- **Full-text, figure, table, or supplementary values** — abstract-only. A number
|
|
180
|
+
that appears only in the body stays `UNVERIFIABLE`.
|
|
181
|
+
- **Relational or qualitative claims** — proportionalities, mechanisms,
|
|
182
|
+
"broader/stronger than". Only value+unit and literal claims are checked.
|
|
183
|
+
- **Papers whose publisher withholds the abstract** — some titles expose no
|
|
184
|
+
abstract to CrossRef or OpenAlex. No abstract → `UNVERIFIABLE`, which reflects
|
|
185
|
+
reachability, not the claim.
|
|
186
|
+
- **Statistical metrics** (p-value, AUC/AUROC, F1, hazard/odds ratio, confidence
|
|
187
|
+
intervals) — handled by the manual skill protocol, not the CLI.
|
|
188
|
+
- **Paper quality, novelty, field consensus**, or whether the *full* paper
|
|
189
|
+
supports a broader statement.
|
|
190
|
+
|
|
191
|
+
**Reading a verdict**
|
|
192
|
+
|
|
193
|
+
| Verdict | Meaning |
|
|
194
|
+
|---|---|
|
|
195
|
+
| `ACCEPT` | The fetched abstract explicitly supports the claim. High-confidence pass. |
|
|
196
|
+
| `WARN` / `PARTIAL` | An abstract was read but does not explicitly support the exact claim. Check the source. |
|
|
197
|
+
| `UNVERIFIABLE` | No abstract was reachable to check against. Not a judgment on the claim. |
|
|
198
|
+
| `REJECT` | DOI is dead, resolves to a different paper, contradicted, or retracted. |
|
|
199
|
+
|
|
200
|
+
---
|
|
201
|
+
|
|
161
202
|
## Modes
|
|
162
203
|
|
|
163
204
|
**Quick Screen** is for DOIs you already have. It uses CrossRef to compare the
|
|
@@ -171,8 +212,8 @@ ref-verify verify-doi <doi> --title "<title>" --first-author <last-name> --year
|
|
|
171
212
|
exit code, so weak or mismatched metadata cannot silently pass automation gates.
|
|
172
213
|
|
|
173
214
|
**Full Audit** is for literature search and final pre-submission review. The
|
|
174
|
-
skill fetches abstracts through CrossRef, Semantic Scholar, Unpaywall,
|
|
175
|
-
and PubMed where needed, then checks whether the paper supports the specific
|
|
215
|
+
skill fetches abstracts through CrossRef, OpenAlex, Semantic Scholar, Unpaywall,
|
|
216
|
+
arXiv, and PubMed where needed, then checks whether the paper supports the specific
|
|
176
217
|
claim being cited.
|
|
177
218
|
|
|
178
219
|
For a single DOI-backed claim, the CLI can run the abstract check:
|
|
@@ -186,6 +227,28 @@ ref-verify check-claim <doi> --claim "<specific claim>" --json
|
|
|
186
227
|
`abstract_source`, `source_attempts`, and `error_code` so agents can distinguish
|
|
187
228
|
missing abstracts, source failures, DOI mismatches, and ambiguous evidence.
|
|
188
229
|
|
|
230
|
+
Use `check-file` when a draft, literature note, or AI-agent output has many
|
|
231
|
+
DOI/claim pairs.
|
|
232
|
+
|
|
233
|
+
JSONL:
|
|
234
|
+
|
|
235
|
+
```bash
|
|
236
|
+
ref-verify check-file claims.jsonl
|
|
237
|
+
ref-verify check-file claims.jsonl --json
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
CSV:
|
|
241
|
+
|
|
242
|
+
```bash
|
|
243
|
+
ref-verify check-file claims.csv
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
Each row must include `doi` and `claim`. Optional fields are `id`, `source`,
|
|
247
|
+
and `note`. Batch mode reuses the same conservative `check-claim` engine:
|
|
248
|
+
`ACCEPT` means the abstract explicitly supports the numeric claim. `WARN`,
|
|
249
|
+
`PARTIAL`, `REJECT`, or `UNVERIFIABLE` means the claim should not be treated as
|
|
250
|
+
verified.
|
|
251
|
+
|
|
189
252
|
Current `check-claim` error codes:
|
|
190
253
|
|
|
191
254
|
- `CLAIM_SUPPORTED`: explicit abstract support found.
|
|
@@ -194,7 +257,7 @@ Current `check-claim` error codes:
|
|
|
194
257
|
- `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
|
|
195
258
|
- `DOI_NOT_FOUND`: selected source did not find a DOI-bound record.
|
|
196
259
|
- `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
|
|
197
|
-
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_UNSUPPORTED`: source lookup failed or could not be used.
|
|
260
|
+
- `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
|
|
198
261
|
|
|
199
262
|
> Core rule: every content statement about a paper must come from a live-fetched
|
|
200
263
|
> abstract. If the abstract is inaccessible after fallback checks, say
|
|
@@ -202,6 +202,8 @@ def _final_error_code(
|
|
|
202
202
|
return "SOURCE_API_ERROR"
|
|
203
203
|
if "TIMEOUT" in statuses:
|
|
204
204
|
return "SOURCE_TIMEOUT"
|
|
205
|
+
if "RATE_LIMITED" in statuses:
|
|
206
|
+
return "SOURCE_RATE_LIMITED"
|
|
205
207
|
if "UNSUPPORTED" in statuses and "NO_ABSTRACT" not in statuses:
|
|
206
208
|
return "SOURCE_UNSUPPORTED"
|
|
207
209
|
if statuses and all(status == "NOT_FOUND" for status in statuses):
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import asdict, dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Literal
|
|
8
|
+
|
|
9
|
+
BatchFormat = Literal["jsonl", "csv"]
|
|
10
|
+
_VALID_SOURCES = {"auto", "crossref", "openalex", "semantic-scholar", "pubmed"}
|
|
11
|
+
_FAILED_ERROR_CODES = {
|
|
12
|
+
"ROW_CHECK_ERROR",
|
|
13
|
+
"SOURCE_API_ERROR",
|
|
14
|
+
"SOURCE_TIMEOUT",
|
|
15
|
+
"SOURCE_RATE_LIMITED",
|
|
16
|
+
"SOURCE_UNSUPPORTED",
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class BatchInputError(ValueError):
|
|
21
|
+
pass
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class ClaimInputRow:
|
|
26
|
+
row_number: int
|
|
27
|
+
id: str | None
|
|
28
|
+
doi: str
|
|
29
|
+
claim: str
|
|
30
|
+
source: str = "auto"
|
|
31
|
+
note: str | None = None
|
|
32
|
+
|
|
33
|
+
def to_dict(self) -> dict[str, Any]:
|
|
34
|
+
return asdict(self)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class BatchRowResult:
|
|
39
|
+
row: ClaimInputRow
|
|
40
|
+
payload: dict[str, Any]
|
|
41
|
+
|
|
42
|
+
def to_dict(self) -> dict[str, Any]:
|
|
43
|
+
result = {
|
|
44
|
+
"row_number": self.row.row_number,
|
|
45
|
+
"id": self.row.id,
|
|
46
|
+
"doi": self.row.doi,
|
|
47
|
+
"claim": self.row.claim,
|
|
48
|
+
}
|
|
49
|
+
if self.row.note is not None:
|
|
50
|
+
result["note"] = self.row.note
|
|
51
|
+
result.update(self.payload)
|
|
52
|
+
return result
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(frozen=True)
|
|
56
|
+
class BatchSummary:
|
|
57
|
+
total: int
|
|
58
|
+
accept: int
|
|
59
|
+
warn: int
|
|
60
|
+
reject: int
|
|
61
|
+
partial: int
|
|
62
|
+
unverifiable: int
|
|
63
|
+
failed: int
|
|
64
|
+
|
|
65
|
+
def to_dict(self) -> dict[str, int]:
|
|
66
|
+
return asdict(self)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def detect_format(path: Path, explicit_format: str | None) -> BatchFormat:
|
|
70
|
+
if explicit_format in ("jsonl", "csv"):
|
|
71
|
+
return explicit_format
|
|
72
|
+
if explicit_format is not None:
|
|
73
|
+
raise BatchInputError(f"Unsupported input format: {explicit_format}")
|
|
74
|
+
suffix = path.suffix.lower()
|
|
75
|
+
if suffix == ".jsonl":
|
|
76
|
+
return "jsonl"
|
|
77
|
+
if suffix == ".csv":
|
|
78
|
+
return "csv"
|
|
79
|
+
raise BatchInputError("Unsupported input format; use .jsonl, .csv, or --format")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def parse_claim_file(path: Path, explicit_format: str | None) -> list[ClaimInputRow]:
|
|
83
|
+
batch_format = detect_format(path, explicit_format)
|
|
84
|
+
try:
|
|
85
|
+
if batch_format == "jsonl":
|
|
86
|
+
rows = _parse_jsonl(path)
|
|
87
|
+
else:
|
|
88
|
+
rows = _parse_csv(path)
|
|
89
|
+
except OSError as exc:
|
|
90
|
+
raise BatchInputError(f"Could not read input file: {exc}") from exc
|
|
91
|
+
if not rows:
|
|
92
|
+
raise BatchInputError("Input file does not contain any claim rows")
|
|
93
|
+
return rows
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def summarize_results(results: list[BatchRowResult]) -> BatchSummary:
|
|
97
|
+
accept = warn = reject = partial = unverifiable = failed = 0
|
|
98
|
+
for result in results:
|
|
99
|
+
verdict = str(result.payload.get("verdict", ""))
|
|
100
|
+
status = str(result.payload.get("status", ""))
|
|
101
|
+
if verdict == "ACCEPT":
|
|
102
|
+
accept += 1
|
|
103
|
+
if verdict == "WARN":
|
|
104
|
+
warn += 1
|
|
105
|
+
if verdict == "REJECT":
|
|
106
|
+
reject += 1
|
|
107
|
+
if status == "PARTIAL":
|
|
108
|
+
partial += 1
|
|
109
|
+
if status == "UNVERIFIABLE":
|
|
110
|
+
unverifiable += 1
|
|
111
|
+
if verdict == "ERROR" or str(result.payload.get("error_code", "")) in _FAILED_ERROR_CODES:
|
|
112
|
+
failed += 1
|
|
113
|
+
return BatchSummary(
|
|
114
|
+
total=len(results),
|
|
115
|
+
accept=accept,
|
|
116
|
+
warn=warn,
|
|
117
|
+
reject=reject,
|
|
118
|
+
partial=partial,
|
|
119
|
+
unverifiable=unverifiable,
|
|
120
|
+
failed=failed,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def batch_payload(results: list[BatchRowResult]) -> dict[str, Any]:
|
|
125
|
+
return {
|
|
126
|
+
"summary": summarize_results(results).to_dict(),
|
|
127
|
+
"results": [result.to_dict() for result in results],
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def render_batch_text(results: list[BatchRowResult]) -> str:
|
|
132
|
+
summary = summarize_results(results)
|
|
133
|
+
lines = [
|
|
134
|
+
(
|
|
135
|
+
f"Summary: total={summary.total} accept={summary.accept} "
|
|
136
|
+
f"warn={summary.warn} reject={summary.reject} "
|
|
137
|
+
f"partial={summary.partial} unverifiable={summary.unverifiable} "
|
|
138
|
+
f"failed={summary.failed}"
|
|
139
|
+
)
|
|
140
|
+
]
|
|
141
|
+
for result in results:
|
|
142
|
+
payload = result.payload
|
|
143
|
+
label = str(payload.get("verdict", "WARN"))
|
|
144
|
+
row_id = result.row.id or f"row-{result.row.row_number}"
|
|
145
|
+
lines.extend(
|
|
146
|
+
[
|
|
147
|
+
"",
|
|
148
|
+
f"{label} {row_id} {result.row.doi}",
|
|
149
|
+
f"Claim: {result.row.claim}",
|
|
150
|
+
f"Reason: {payload.get('reason', '')}",
|
|
151
|
+
]
|
|
152
|
+
)
|
|
153
|
+
evidence = payload.get("evidence")
|
|
154
|
+
if evidence:
|
|
155
|
+
lines.append(f"Evidence: {evidence}")
|
|
156
|
+
error_code = payload.get("error_code")
|
|
157
|
+
if error_code:
|
|
158
|
+
lines.append(f"Error code: {error_code}")
|
|
159
|
+
return "\n".join(lines)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _parse_jsonl(path: Path) -> list[ClaimInputRow]:
|
|
163
|
+
rows: list[ClaimInputRow] = []
|
|
164
|
+
with path.open("r", encoding="utf-8") as handle:
|
|
165
|
+
for line_number, line in enumerate(handle, start=1):
|
|
166
|
+
stripped = line.strip()
|
|
167
|
+
if not stripped:
|
|
168
|
+
continue
|
|
169
|
+
try:
|
|
170
|
+
raw = json.loads(stripped)
|
|
171
|
+
except json.JSONDecodeError as exc:
|
|
172
|
+
raise BatchInputError(f"Invalid JSON on line {line_number}: {exc.msg}") from exc
|
|
173
|
+
if not isinstance(raw, dict):
|
|
174
|
+
raise BatchInputError(f"Invalid row on line {line_number}: expected object")
|
|
175
|
+
rows.append(_row_from_mapping(raw, line_number=line_number, row_label="line"))
|
|
176
|
+
return rows
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _parse_csv(path: Path) -> list[ClaimInputRow]:
|
|
180
|
+
rows: list[ClaimInputRow] = []
|
|
181
|
+
with path.open("r", encoding="utf-8-sig", newline="") as handle:
|
|
182
|
+
reader = csv.DictReader(handle, skipinitialspace=True)
|
|
183
|
+
if reader.fieldnames is None:
|
|
184
|
+
raise BatchInputError("CSV input is missing a header row")
|
|
185
|
+
fieldnames = {_normalize_csv_fieldname(field) for field in reader.fieldnames if field}
|
|
186
|
+
missing = {"doi", "claim"} - fieldnames
|
|
187
|
+
if missing:
|
|
188
|
+
missing_fields = ", ".join(sorted(missing))
|
|
189
|
+
raise BatchInputError(f"CSV header must include doi and claim fields; missing: {missing_fields}")
|
|
190
|
+
for row_number, raw in enumerate(reader, start=2):
|
|
191
|
+
rows.append(
|
|
192
|
+
_row_from_mapping(
|
|
193
|
+
_normalize_csv_row(raw),
|
|
194
|
+
line_number=row_number,
|
|
195
|
+
row_label="line",
|
|
196
|
+
)
|
|
197
|
+
)
|
|
198
|
+
return rows
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _row_from_mapping(raw: dict[str, Any], *, line_number: int, row_label: str) -> ClaimInputRow:
|
|
202
|
+
doi = _required_string(raw, "doi", line_number, row_label)
|
|
203
|
+
claim = _required_string(raw, "claim", line_number, row_label)
|
|
204
|
+
source = _optional_string(raw, "source") or "auto"
|
|
205
|
+
if source not in _VALID_SOURCES:
|
|
206
|
+
raise BatchInputError(f"Invalid field on {row_label} {line_number}: source={source}")
|
|
207
|
+
return ClaimInputRow(
|
|
208
|
+
row_number=line_number,
|
|
209
|
+
id=_optional_string(raw, "id"),
|
|
210
|
+
doi=doi,
|
|
211
|
+
claim=claim,
|
|
212
|
+
source=source,
|
|
213
|
+
note=_optional_string(raw, "note"),
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _required_string(raw: dict[str, Any], field: str, line_number: int, row_label: str) -> str:
|
|
218
|
+
value = raw.get(field)
|
|
219
|
+
if not isinstance(value, str) or not value.strip():
|
|
220
|
+
raise BatchInputError(f"Missing required field on {row_label} {line_number}: {field}")
|
|
221
|
+
return value.strip()
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _optional_string(raw: dict[str, Any], field: str) -> str | None:
|
|
225
|
+
value = raw.get(field)
|
|
226
|
+
if value is None:
|
|
227
|
+
return None
|
|
228
|
+
if not isinstance(value, str):
|
|
229
|
+
return str(value)
|
|
230
|
+
stripped = value.strip()
|
|
231
|
+
return stripped or None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _normalize_csv_fieldname(field: str) -> str:
|
|
235
|
+
return field.lstrip("\ufeff").strip()
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _normalize_csv_row(raw: dict[str, Any]) -> dict[str, Any]:
|
|
239
|
+
return {_normalize_csv_fieldname(key): value for key, value in raw.items() if key is not None}
|
|
@@ -209,7 +209,7 @@ _UNSUPPORTED_CLAIM_FRAME_PATTERNS = (
|
|
|
209
209
|
r"\bsaid to\b",
|
|
210
210
|
r"\b(?:expect|expects|expected|expecting) to\b",
|
|
211
211
|
r"\b(?:project|projects|projected|projecting) to\b",
|
|
212
|
-
r"\b(?:estimate|estimates|estimated|estimating) to\b",
|
|
212
|
+
r"\b(?:estimate|estimates|estimated|estimating) to (?!be\b|equal\b|at\b|about\b)",
|
|
213
213
|
r"\bachievable\b",
|
|
214
214
|
r"\bpossible\b",
|
|
215
215
|
)
|
|
@@ -338,6 +338,8 @@ def _claim_percentage_comparator(claim: str) -> str:
|
|
|
338
338
|
return "gte"
|
|
339
339
|
if re.search(r"\b(at most|no more than)\b", normalized):
|
|
340
340
|
return "lte"
|
|
341
|
+
if re.search(r"\bup to\b", normalized):
|
|
342
|
+
return "up_to"
|
|
341
343
|
if re.search(r"\b(below|under|less than)\b", normalized):
|
|
342
344
|
return "lt"
|
|
343
345
|
if re.search(
|
|
@@ -414,6 +416,8 @@ def _evidence_entails_claim(
|
|
|
414
416
|
claim_comparator: str,
|
|
415
417
|
) -> bool:
|
|
416
418
|
if evidence_comparator == "up_to":
|
|
419
|
+
if claim_comparator == "up_to":
|
|
420
|
+
return value == threshold
|
|
417
421
|
if claim_comparator == "exact":
|
|
418
422
|
return False
|
|
419
423
|
return _compare_percentage(value, threshold, claim_comparator)
|