ref-verify 1.2.2__tar.gz → 1.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {ref_verify-1.2.2/src/ref_verify.egg-info → ref_verify-1.3.1}/PKG-INFO +184 -3
  2. {ref_verify-1.2.2 → ref_verify-1.3.1}/README.md +183 -2
  3. {ref_verify-1.2.2 → ref_verify-1.3.1}/pyproject.toml +1 -1
  4. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/__init__.py +1 -1
  5. ref_verify-1.3.1/src/ref_verify/cache.py +90 -0
  6. ref_verify-1.3.1/src/ref_verify/cli.py +520 -0
  7. ref_verify-1.3.1/src/ref_verify/crossref.py +256 -0
  8. ref_verify-1.3.1/src/ref_verify/doi_check.py +324 -0
  9. ref_verify-1.3.1/src/ref_verify/http.py +121 -0
  10. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/models.py +14 -1
  11. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/openalex.py +16 -16
  12. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/pubmed.py +15 -26
  13. ref_verify-1.3.1/src/ref_verify/reference_parse.py +462 -0
  14. ref_verify-1.3.1/src/ref_verify/reference_resolve.py +625 -0
  15. ref_verify-1.3.1/src/ref_verify/report.py +318 -0
  16. ref_verify-1.3.1/src/ref_verify/semantic_scholar.py +110 -0
  17. {ref_verify-1.2.2 → ref_verify-1.3.1/src/ref_verify.egg-info}/PKG-INFO +184 -3
  18. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify.egg-info/SOURCES.txt +10 -0
  19. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_abstract_sources.py +26 -6
  20. ref_verify-1.3.1/tests/test_benchmark_aggregate.py +76 -0
  21. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_cli.py +225 -1
  22. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_crossref.py +12 -0
  23. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_doi_check.py +22 -1
  24. ref_verify-1.3.1/tests/test_http_cache.py +452 -0
  25. ref_verify-1.3.1/tests/test_reference_parse.py +210 -0
  26. ref_verify-1.3.1/tests/test_reference_resolve.py +1059 -0
  27. ref_verify-1.3.1/tests/test_report.py +324 -0
  28. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_skill_docs.py +16 -7
  29. ref_verify-1.2.2/src/ref_verify/cli.py +0 -280
  30. ref_verify-1.2.2/src/ref_verify/crossref.py +0 -104
  31. ref_verify-1.2.2/src/ref_verify/doi_check.py +0 -208
  32. ref_verify-1.2.2/src/ref_verify/semantic_scholar.py +0 -107
  33. {ref_verify-1.2.2 → ref_verify-1.3.1}/LICENSE +0 -0
  34. {ref_verify-1.2.2 → ref_verify-1.3.1}/setup.cfg +0 -0
  35. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/abstract_lookup.py +0 -0
  36. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/batch.py +0 -0
  37. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/claim_check.py +0 -0
  38. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify/numeric_claim.py +0 -0
  39. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify.egg-info/dependency_links.txt +0 -0
  40. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify.egg-info/entry_points.txt +0 -0
  41. {ref_verify-1.2.2 → ref_verify-1.3.1}/src/ref_verify.egg-info/top_level.txt +0 -0
  42. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_batch.py +0 -0
  43. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_claim_check.py +0 -0
  44. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_numeric_claim.py +0 -0
  45. {ref_verify-1.2.2 → ref_verify-1.3.1}/tests/test_package_smoke.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ref-verify
3
- Version: 1.2.2
3
+ Version: 1.3.1
4
4
  Summary: Executable DOI and claim verification helpers for academic citations
5
5
  Author: Moonweave Research
6
6
  License-Expression: MIT
@@ -30,6 +30,44 @@ supports a specific claim, or audit references before submission. No server setu
30
30
 
31
31
  ---
32
32
 
33
+ ## Scorecard
34
+
35
+ <picture>
36
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/Moonweave-Research/ref-verify/main/.github/assets/scorecard-dark.svg">
37
+ <img src="https://raw.githubusercontent.com/Moonweave-Research/ref-verify/main/.github/assets/scorecard-light.svg" alt="Bar chart of check-bib verdicts on 86 held-out references: real 80% passed cleanly, fabricated 96% flagged, retracted 100% caught, 0 of 10 unindexed references rejected." width="830">
38
+ </picture>
39
+
40
+ Held-out set: 86 references written and committed before the tool was run on them, with no paper
41
+ shared with the development set.
42
+
43
+ | What was measured (held-out set) | Result | n | 95% CI |
44
+ |---|---|---|---|
45
+ | Real papers passed cleanly | **80%** | 40 | 65–90% |
46
+ | Real papers sent for a manual check (WARN) | 20% | 40 | 10–35% |
47
+ | Real papers wrongly rejected | 0% | 40 | 0–9% |
48
+ | Fabricated references flagged (WARN or REJECT) | **96%** | 26 | 81–99% |
49
+ | Retracted papers caught as `PAPER_RETRACTED` | **100%** | 10 | 72–100% |
50
+ | Legitimate references missing from CrossRef that were rejected | 0 of 10 | 10 | 0–28% |
51
+
52
+ - Fabricated, by type: invented DOI 5/5 · no DOI 5/5 · DOI swap 4/4 · wrong author/year 4/5 · publicly reported cases 7/7.
53
+ - 7 of the 8 real papers that did not pass are cited in the physics/chemistry style that omits the
54
+ article title, which leaves nothing to compare against CrossRef.
55
+ - Time for all 86 references: 79 s on a cold cache (1.1 s median per reference), 0.1 s cached.
56
+
57
+ Development set (142 references, used while fixing the tool in
58
+ [#27](https://github.com/Moonweave-Research/ref-verify/pull/27), so these are in-sample scores): real
59
+ 66/66 passed (94–100%), fabricated 43/43 flagged, retracted 16/16 caught, 1 of
60
+ 17 unindexed references rejected.
61
+
62
+ Measured 2026-10-08 with ref-verify 1.2.2 (commit `01c7a37`) against live CrossRef, checking each set with
63
+ `check-bib` as BibTeX, RIS, and plain-text lists. Not measured: whether a paper supports a claim
64
+ (beyond a small numeric fixture), non-English literature beyond a few Korean items, and full text.
65
+ Every miss is listed per item in the results files ([held-out](https://github.com/Moonweave-Research/ref-verify/blob/main/benchmarks/results/2026-10-08-01c7a37-holdout-v1.json),
66
+ [development](https://github.com/Moonweave-Research/ref-verify/blob/main/benchmarks/results/2026-10-08-01c7a37-v1.json)); dataset, method, and how to rerun:
67
+ [benchmarks/README.md](https://github.com/Moonweave-Research/ref-verify/blob/main/benchmarks/README.md).
68
+
69
+ ---
70
+
33
71
  ## Install the skill
34
72
 
35
73
  ```bash
@@ -68,6 +106,61 @@ formatting, and citation style questions.
68
106
 
69
107
  ---
70
108
 
109
+ ## Check a whole reference list
110
+
111
+ Find references in a paper or thesis that do not exist (for example ones a
112
+ chatbot made up), whose DOI points to a different paper, or that were
113
+ retracted, in one run.
114
+
115
+ **With the agent:** after installing the skill, ask "check every reference in
116
+ references.bib with ref-verify".
117
+
118
+ **From a terminal:**
119
+
120
+ 1. Put the list in a file.
121
+ - Zotero: right-click the collection → Export Collection → BibTeX →
122
+ `references.bib` (EndNote and Mendeley export BibTeX or RIS).
123
+ - A Word or other manuscript: copy the reference list into a plain-text
124
+ editor and save it as `references.txt`. `[1]` or `1.` numbering and
125
+ wrapped lines are fine. `.docx` and `.pdf` files are not read directly.
126
+ 2. Install (Python 3.10 or newer):
127
+
128
+ ```bash
129
+ pipx install ref-verify
130
+ ```
131
+
132
+ With `uv`, skip the install:
133
+ `uvx ref-verify check-bib references.bib`.
134
+
135
+ 3. Run:
136
+
137
+ ```bash
138
+ ref-verify check-bib references.bib
139
+ ```
140
+
141
+ A first run takes about a second per reference (a little over two minutes
142
+ for 150, with a `Checking references: 37/150` counter). Running the same
143
+ list again takes seconds thanks to the cache. Prefixing
144
+ `REF_VERIFY_MAILTO=you@university.edu` uses CrossRef's polite pool and is
145
+ about three times faster.
146
+
147
+ Add `--report check.html` for a file to send to an advisor or co-author;
148
+ it opens in a browser with the items that need a look at the top.
149
+
150
+ **Reading the result**
151
+
152
+ | Result | Meaning | What to do |
153
+ |---|---|---|
154
+ | `PASS` | Title, first author, and year match the CrossRef record for the DOI (or the record found by search) | Nothing |
155
+ | `WARN` | Found, but something differs; the line below says what (year, author, the title of the paper the DOI really points to) | Compare that one with the source |
156
+ | `REJECT` | The DOI exists nowhere, points to a different paper, or the paper is retracted | Fix or drop the citation |
157
+ | `UNVERIFIED` | Could not be confirmed automatically; theses, local conference abstracts, some books, and DOIs registered outside CrossRef (arXiv, KISTI) often land here. It does not mean the reference is wrong | Check it yourself |
158
+
159
+ A made-up reference without a DOI can only show as `UNVERIFIED`, not `REJECT`,
160
+ so look each `UNVERIFIED` item up once (for example in Google Scholar).
161
+
162
+ ---
163
+
71
164
  ## Optional CLI engine
72
165
 
73
166
  The skill is the agent workflow. The Python CLI is the skill-level execution engine that the installed skill can call from a terminal.
@@ -84,6 +177,7 @@ checks that are currently safe to automate directly:
84
177
  - subject-matched percentage claims such as efficiency, response rate, or actuation strain
85
178
  - simple unit/count claims such as cycles, patients, voltage, temperature, and concentration
86
179
  - CrossRef first, then DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback when CrossRef has no abstract
180
+ - Reference-list check (BibTeX, RIS, plain text, Markdown): `ref-verify check-bib`
87
181
  - JSON output for agent-readable routing
88
182
  - Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
89
183
 
@@ -93,6 +187,18 @@ The CLI has zero third-party Python runtime dependencies, but it is not an
93
187
  offline verifier. Functional checks require outbound HTTPS access to public
94
188
  academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
95
189
 
190
+ ### Cache
191
+
192
+ The CLI keeps API responses on disk for 7 days, so re-running a check does not
193
+ query CrossRef and the abstract sources again. A DOI that returned HTTP 404 is
194
+ kept for 1 day only, so a newly registered DOI is re-checked soon. Rate limits
195
+ (429) and server errors (5xx) are retried up to 3 times with backoff, honouring
196
+ `Retry-After` up to 10 s, and are never cached.
197
+
198
+ - Location: `$REF_VERIFY_CACHE_DIR`, else `$XDG_CACHE_HOME/ref-verify`, else `~/.cache/ref-verify`.
199
+ - Lifetime: `REF_VERIFY_CACHE_TTL_DAYS` (default `7`).
200
+ - Disable: `--no-cache` on any command, or `REF_VERIFY_NO_CACHE=1`. Delete the directory to clear it.
201
+
96
202
  To run the CLI yourself, install it from PyPI:
97
203
 
98
204
  ```bash
@@ -271,7 +377,11 @@ ref-verify check-file claims.csv
271
377
  ```
272
378
 
273
379
  Each row must include `doi` and `claim`. Optional fields are `id`, `source`,
274
- and `note`. Batch mode reuses the same conservative `check-claim` engine:
380
+ and `note`. Rows are checked 4 at a time by default (`--workers N`); output keeps
381
+ the input order, and CrossRef and Semantic Scholar requests go one at a time
382
+ because their public APIs reject parallel requests. In a terminal, a
383
+ `Checking claims: N/M` counter on stderr shows progress (never with `--json`).
384
+ Ctrl-C stops the run; finished lookups stay cached, so rerunning resumes quickly. Batch mode reuses the same conservative `check-claim` engine:
275
385
  `ACCEPT` means the abstract explicitly supports the numeric claim. `WARN`,
276
386
  `PARTIAL`, `REJECT`, or `UNVERIFIABLE` means the claim should not be treated as
277
387
  verified.
@@ -282,11 +392,82 @@ Current `check-claim` error codes:
282
392
  - `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
283
393
  - `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
284
394
  - `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
285
- - `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
395
+ - `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI, or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
396
+ - `DOI_NOT_IN_CROSSREF`: CrossRef has no record, but doi.org lists the DOI with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...). `verify-doi` returns `verdict: WARN`, `status: UNVERIFIED` without comparing metadata; `check-claim` tries OpenAlex, Semantic Scholar (arXiv DOIs by arXiv identifier), and PubMed for the abstract and judges the claim if one has it, otherwise returns `status: UNVERIFIABLE`, `verdict: WARN` with this code. Not a dead DOI.
286
397
  - `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
287
398
  - `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
288
399
  - `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
289
400
 
401
+ Use `check-bib` when you have a reference list rather than DOI/claim pairs:
402
+
403
+ ```bash
404
+ ref-verify check-bib references.bib
405
+ ref-verify check-bib references.ris --json
406
+ ref-verify check-bib references.md --format txt
407
+ ```
408
+
409
+ It reads BibTeX, RIS, and plain-text or Markdown lists (one reference per
410
+ paragraph, per line, or per `[1]`/`1.`/`1)` item). A reference with a DOI is
411
+ compared with its CrossRef record like `verify-doi`; a plain-text reference
412
+ passes only when its text shows the CrossRef title and first author. A
413
+ reference without a DOI is looked up with CrossRef bibliographic search and
414
+ accepted only when the title matches and the year is within one. Matching
415
+ accepts the print or the online-first year, a title with or without its
416
+ subtitle or edition note, TeX math in BibTeX titles (`$\beta$` reads as β),
417
+ CrossRef's original-language title (for example the Korean title of
418
+ a *Polymer Korea* paper), and Hangul author names against CrossRef's
419
+ romanized ones (윤 → Yoon/Yun). When a DOI is unknown to CrossRef, doi.org is
420
+ asked which agency registered it, so arXiv, Zenodo, or KISTI DOIs are not
421
+ reported as dead. Search results that are about the paper rather than the
422
+ paper itself (peer-review reports, Faculty Opinions recommendations,
423
+ addenda and corrections) are skipped. A citation in a style that omits the
424
+ article title (`J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108,
425
+ 1175 (1957)`) is compared on journal (full name or abbreviation), volume, first
426
+ page or article number, year, and first author; it passes when all of them
427
+ agree, and otherwise the reason names each field that differs. Without a DOI,
428
+ when the plain search finds nothing and the citation looks title-less, a second
429
+ CrossRef search by first author,
430
+ the rest of the citation, and the cited year finds short citations such as
431
+ `A. G. Riess et al., Astron. J. 116, 1009 (1998).`; the same full agreement is
432
+ required. The first author
433
+ is read only from the first name in the list, so a reference that puts a
434
+ co-author first does not pass. The terminal output starts with a count line
435
+ (`19 references: 11 PASS, 2 WARN, 5 REJECT, 1 UNVERIFIED`), lists one row per
436
+ reference (citation key, or the start of the reference for a pasted list), puts
437
+ the reason under every row that is not `PASS`, and ends with a one-paragraph
438
+ legend. With `--json` it is an object with `summary` (`total`, `pass`, `warn`,
439
+ `reject`, `unverified`, `failed`; `warn` includes the `UNVERIFIED` rows) and
440
+ `results`. `check-bib` exits `0` only
441
+ when every reference is `PASS`.
442
+
443
+ `check-bib` error codes:
444
+
445
+ - `REFERENCE_RESOLVED`: the reference had no DOI; CrossRef search found a matching record, reported as `resolved_doi`. `WARN` when the year differs by one or the first author differs.
446
+ - `REFERENCE_UNMATCHED`: the reference had no DOI and no CrossRef record matched (`status: UNVERIFIED`, `verdict: WARN`). The tool could not confirm it automatically; that does not mean the reference is wrong. Verify it manually.
447
+ - `DOI_NOT_IN_CROSSREF`: the DOI is registered with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...), so its metadata was not compared (`status: UNVERIFIED`, `verdict: WARN`). Open the DOI to confirm it.
448
+ - `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI (`REJECT`).
449
+ - `PAPER_RETRACTED`, `ROW_CHECK_ERROR`: as for `check-claim` and `check-file`. Other DOI-backed results carry `error_code: null`; read `verdict`, `mismatches`, and `reason`, which names what differs (for example `the year differs (reference: 2009; CrossRef: 2010)`). A plain-text reference whose DOI belongs to a paper it does not mention is `status: MISMATCH`, `verdict: WARN`, with that paper's title in `reason`.
450
+
451
+ To hand the result to a co-author or supervisor, add `--report` to `check-bib`
452
+ or `check-file`. The file extension picks the format:
453
+
454
+ ```bash
455
+ ref-verify check-bib references.bib --report report.html
456
+ ref-verify check-file claims.jsonl --report report.md
457
+ ```
458
+
459
+ The HTML file is self-contained (inline CSS, no scripts, no external resources
460
+ other than `https://doi.org/` links). It opens with counts that add up to the
461
+ total (one box per verdict as shown), a plain-language line on what each
462
+ verdict means and asks you to do, then a "Needs a look" table with every
463
+ non-passing reference or claim and a "Passed" table below it. Each row is
464
+ coloured (`PASS`/`ACCEPT` green, `WARN` amber, `REJECT` red, `UNVERIFIED`
465
+ grey) and shows the reason and evidence. `UNVERIFIED` marks a result the tool
466
+ could not confirm automatically; it is not a finding that the reference is
467
+ wrong. The Markdown file has the same content. A `--report` path whose folder
468
+ does not exist is rejected before any lookup, so a long run is never lost.
469
+ `--json` output is unchanged.
470
+
290
471
  > Core rule: every content statement about a paper must come from a live-fetched
291
472
  > source at the depth the claim requires — abstract for topline claims, full
292
473
  > text for mechanism, implementation, or procedural claims. If the required
@@ -19,6 +19,44 @@ supports a specific claim, or audit references before submission. No server setu
19
19
 
20
20
  ---
21
21
 
22
+ ## Scorecard
23
+
24
+ <picture>
25
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/Moonweave-Research/ref-verify/main/.github/assets/scorecard-dark.svg">
26
+ <img src="https://raw.githubusercontent.com/Moonweave-Research/ref-verify/main/.github/assets/scorecard-light.svg" alt="Bar chart of check-bib verdicts on 86 held-out references: real 80% passed cleanly, fabricated 96% flagged, retracted 100% caught, 0 of 10 unindexed references rejected." width="830">
27
+ </picture>
28
+
29
+ Held-out set: 86 references written and committed before the tool was run on them, with no paper
30
+ shared with the development set.
31
+
32
+ | What was measured (held-out set) | Result | n | 95% CI |
33
+ |---|---|---|---|
34
+ | Real papers passed cleanly | **80%** | 40 | 65–90% |
35
+ | Real papers sent for a manual check (WARN) | 20% | 40 | 10–35% |
36
+ | Real papers wrongly rejected | 0% | 40 | 0–9% |
37
+ | Fabricated references flagged (WARN or REJECT) | **96%** | 26 | 81–99% |
38
+ | Retracted papers caught as `PAPER_RETRACTED` | **100%** | 10 | 72–100% |
39
+ | Legitimate references missing from CrossRef that were rejected | 0 of 10 | 10 | 0–28% |
40
+
41
+ - Fabricated, by type: invented DOI 5/5 · no DOI 5/5 · DOI swap 4/4 · wrong author/year 4/5 · publicly reported cases 7/7.
42
+ - 7 of the 8 real papers that did not pass are cited in the physics/chemistry style that omits the
43
+ article title, which leaves nothing to compare against CrossRef.
44
+ - Time for all 86 references: 79 s on a cold cache (1.1 s median per reference), 0.1 s cached.
45
+
46
+ Development set (142 references, used while fixing the tool in
47
+ [#27](https://github.com/Moonweave-Research/ref-verify/pull/27), so these are in-sample scores): real
48
+ 66/66 passed (94–100%), fabricated 43/43 flagged, retracted 16/16 caught, 1 of
49
+ 17 unindexed references rejected.
50
+
51
+ Measured 2026-10-08 with ref-verify 1.2.2 (commit `01c7a37`) against live CrossRef, checking each set with
52
+ `check-bib` as BibTeX, RIS, and plain-text lists. Not measured: whether a paper supports a claim
53
+ (beyond a small numeric fixture), non-English literature beyond a few Korean items, and full text.
54
+ Every miss is listed per item in the results files ([held-out](https://github.com/Moonweave-Research/ref-verify/blob/main/benchmarks/results/2026-10-08-01c7a37-holdout-v1.json),
55
+ [development](https://github.com/Moonweave-Research/ref-verify/blob/main/benchmarks/results/2026-10-08-01c7a37-v1.json)); dataset, method, and how to rerun:
56
+ [benchmarks/README.md](https://github.com/Moonweave-Research/ref-verify/blob/main/benchmarks/README.md).
57
+
58
+ ---
59
+
22
60
  ## Install the skill
23
61
 
24
62
  ```bash
@@ -57,6 +95,61 @@ formatting, and citation style questions.
57
95
 
58
96
  ---
59
97
 
98
+ ## Check a whole reference list
99
+
100
+ Find references in a paper or thesis that do not exist (for example ones a
101
+ chatbot made up), whose DOI points to a different paper, or that were
102
+ retracted, in one run.
103
+
104
+ **With the agent:** after installing the skill, ask "check every reference in
105
+ references.bib with ref-verify".
106
+
107
+ **From a terminal:**
108
+
109
+ 1. Put the list in a file.
110
+ - Zotero: right-click the collection → Export Collection → BibTeX →
111
+ `references.bib` (EndNote and Mendeley export BibTeX or RIS).
112
+ - A Word or other manuscript: copy the reference list into a plain-text
113
+ editor and save it as `references.txt`. `[1]` or `1.` numbering and
114
+ wrapped lines are fine. `.docx` and `.pdf` files are not read directly.
115
+ 2. Install (Python 3.10 or newer):
116
+
117
+ ```bash
118
+ pipx install ref-verify
119
+ ```
120
+
121
+ With `uv`, skip the install:
122
+ `uvx ref-verify check-bib references.bib`.
123
+
124
+ 3. Run:
125
+
126
+ ```bash
127
+ ref-verify check-bib references.bib
128
+ ```
129
+
130
+ A first run takes about a second per reference (a little over two minutes
131
+ for 150, with a `Checking references: 37/150` counter). Running the same
132
+ list again takes seconds thanks to the cache. Prefixing
133
+ `REF_VERIFY_MAILTO=you@university.edu` uses CrossRef's polite pool and is
134
+ about three times faster.
135
+
136
+ Add `--report check.html` for a file to send to an advisor or co-author;
137
+ it opens in a browser with the items that need a look at the top.
138
+
139
+ **Reading the result**
140
+
141
+ | Result | Meaning | What to do |
142
+ |---|---|---|
143
+ | `PASS` | Title, first author, and year match the CrossRef record for the DOI (or the record found by search) | Nothing |
144
+ | `WARN` | Found, but something differs; the line below says what (year, author, the title of the paper the DOI really points to) | Compare that one with the source |
145
+ | `REJECT` | The DOI exists nowhere, points to a different paper, or the paper is retracted | Fix or drop the citation |
146
+ | `UNVERIFIED` | Could not be confirmed automatically; theses, local conference abstracts, some books, and DOIs registered outside CrossRef (arXiv, KISTI) often land here. It does not mean the reference is wrong | Check it yourself |
147
+
148
+ A made-up reference without a DOI can only show as `UNVERIFIED`, not `REJECT`,
149
+ so look each `UNVERIFIED` item up once (for example in Google Scholar).
150
+
151
+ ---
152
+
60
153
  ## Optional CLI engine
61
154
 
62
155
  The skill is the agent workflow. The Python CLI is the skill-level execution engine that the installed skill can call from a terminal.
@@ -73,6 +166,7 @@ checks that are currently safe to automate directly:
73
166
  - subject-matched percentage claims such as efficiency, response rate, or actuation strain
74
167
  - simple unit/count claims such as cycles, patients, voltage, temperature, and concentration
75
168
  - CrossRef first, then DOI-bound OpenAlex, Semantic Scholar, and PubMed fallback when CrossRef has no abstract
169
+ - Reference-list check (BibTeX, RIS, plain text, Markdown): `ref-verify check-bib`
76
170
  - JSON output for agent-readable routing
77
171
  - Non-zero exit codes for `WARN`, `REJECT`, and `UNVERIFIABLE` results
78
172
 
@@ -82,6 +176,18 @@ The CLI has zero third-party Python runtime dependencies, but it is not an
82
176
  offline verifier. Functional checks require outbound HTTPS access to public
83
177
  academic APIs such as CrossRef, OpenAlex, Semantic Scholar, and PubMed.
84
178
 
179
+ ### Cache
180
+
181
+ The CLI keeps API responses on disk for 7 days, so re-running a check does not
182
+ query CrossRef and the abstract sources again. A DOI that returned HTTP 404 is
183
+ kept for 1 day only, so a newly registered DOI is re-checked soon. Rate limits
184
+ (429) and server errors (5xx) are retried up to 3 times with backoff, honouring
185
+ `Retry-After` up to 10 s, and are never cached.
186
+
187
+ - Location: `$REF_VERIFY_CACHE_DIR`, else `$XDG_CACHE_HOME/ref-verify`, else `~/.cache/ref-verify`.
188
+ - Lifetime: `REF_VERIFY_CACHE_TTL_DAYS` (default `7`).
189
+ - Disable: `--no-cache` on any command, or `REF_VERIFY_NO_CACHE=1`. Delete the directory to clear it.
190
+
85
191
  To run the CLI yourself, install it from PyPI:
86
192
 
87
193
  ```bash
@@ -260,7 +366,11 @@ ref-verify check-file claims.csv
260
366
  ```
261
367
 
262
368
  Each row must include `doi` and `claim`. Optional fields are `id`, `source`,
263
- and `note`. Batch mode reuses the same conservative `check-claim` engine:
369
+ and `note`. Rows are checked 4 at a time by default (`--workers N`); output keeps
370
+ the input order, and CrossRef and Semantic Scholar requests go one at a time
371
+ because their public APIs reject parallel requests. In a terminal, a
372
+ `Checking claims: N/M` counter on stderr shows progress (never with `--json`).
373
+ Ctrl-C stops the run; finished lookups stay cached, so rerunning resumes quickly. Batch mode reuses the same conservative `check-claim` engine:
264
374
  `ACCEPT` means the abstract explicitly supports the numeric claim. `WARN`,
265
375
  `PARTIAL`, `REJECT`, or `UNVERIFIABLE` means the claim should not be treated as
266
376
  verified.
@@ -271,11 +381,82 @@ Current `check-claim` error codes:
271
381
  - `CLAIM_NOT_EXPLICIT`: an abstract was available, but the claim was not explicitly supported.
272
382
  - `CLAIM_AMBIGUOUS`: numeric evidence or context exists, but binding is ambiguous.
273
383
  - `NO_ABSTRACT`: attempted DOI-bound sources did not provide abstract text.
274
- - `DOI_NOT_FOUND`: CrossRef has no record for the DOI (HTTP 404), or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
384
+ - `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI, or the selected source did not find a DOI-bound record. The JSON still carries a `verdict` of `REJECT`.
385
+ - `DOI_NOT_IN_CROSSREF`: CrossRef has no record, but doi.org lists the DOI with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...). `verify-doi` returns `verdict: WARN`, `status: UNVERIFIED` without comparing metadata; `check-claim` tries OpenAlex, Semantic Scholar (arXiv DOIs by arXiv identifier), and PubMed for the abstract and judges the claim if one has it, otherwise returns `status: UNVERIFIABLE`, `verdict: WARN` with this code. Not a dead DOI.
275
386
  - `PAPER_RETRACTED`: CrossRef lists a retraction notice for the DOI; the claim is rejected before any abstract is read.
276
387
  - `DOI_MISMATCH`: the primary or explicitly selected DOI-bound record did not match the requested DOI.
277
388
  - `SOURCE_API_ERROR`, `SOURCE_TIMEOUT`, `SOURCE_RATE_LIMITED`, `SOURCE_UNSUPPORTED`: source lookup failed, timed out, was rate-limited, or could not be used.
278
389
 
390
+ Use `check-bib` when you have a reference list rather than DOI/claim pairs:
391
+
392
+ ```bash
393
+ ref-verify check-bib references.bib
394
+ ref-verify check-bib references.ris --json
395
+ ref-verify check-bib references.md --format txt
396
+ ```
397
+
398
+ It reads BibTeX, RIS, and plain-text or Markdown lists (one reference per
399
+ paragraph, per line, or per `[1]`/`1.`/`1)` item). A reference with a DOI is
400
+ compared with its CrossRef record like `verify-doi`; a plain-text reference
401
+ passes only when its text shows the CrossRef title and first author. A
402
+ reference without a DOI is looked up with CrossRef bibliographic search and
403
+ accepted only when the title matches and the year is within one. Matching
404
+ accepts the print or the online-first year, a title with or without its
405
+ subtitle or edition note, TeX math in BibTeX titles (`$\beta$` reads as β),
406
+ CrossRef's original-language title (for example the Korean title of
407
+ a *Polymer Korea* paper), and Hangul author names against CrossRef's
408
+ romanized ones (윤 → Yoon/Yun). When a DOI is unknown to CrossRef, doi.org is
409
+ asked which agency registered it, so arXiv, Zenodo, or KISTI DOIs are not
410
+ reported as dead. Search results that are about the paper rather than the
411
+ paper itself (peer-review reports, Faculty Opinions recommendations,
412
+ addenda and corrections) are skipped. A citation in a style that omits the
413
+ article title (`J. Bardeen, L. N. Cooper, and J. R. Schrieffer, Phys. Rev. 108,
414
+ 1175 (1957)`) is compared on journal (full name or abbreviation), volume, first
415
+ page or article number, year, and first author; it passes when all of them
416
+ agree, and otherwise the reason names each field that differs. Without a DOI,
417
+ when the plain search finds nothing and the citation looks title-less, a second
418
+ CrossRef search by first author,
419
+ the rest of the citation, and the cited year finds short citations such as
420
+ `A. G. Riess et al., Astron. J. 116, 1009 (1998).`; the same full agreement is
421
+ required. The first author
422
+ is read only from the first name in the list, so a reference that puts a
423
+ co-author first does not pass. The terminal output starts with a count line
424
+ (`19 references: 11 PASS, 2 WARN, 5 REJECT, 1 UNVERIFIED`), lists one row per
425
+ reference (citation key, or the start of the reference for a pasted list), puts
426
+ the reason under every row that is not `PASS`, and ends with a one-paragraph
427
+ legend. With `--json` it is an object with `summary` (`total`, `pass`, `warn`,
428
+ `reject`, `unverified`, `failed`; `warn` includes the `UNVERIFIED` rows) and
429
+ `results`. `check-bib` exits `0` only
430
+ when every reference is `PASS`.
431
+
432
+ `check-bib` error codes:
433
+
434
+ - `REFERENCE_RESOLVED`: the reference had no DOI; CrossRef search found a matching record, reported as `resolved_doi`. `WARN` when the year differs by one or the first author differs.
435
+ - `REFERENCE_UNMATCHED`: the reference had no DOI and no CrossRef record matched (`status: UNVERIFIED`, `verdict: WARN`). The tool could not confirm it automatically; that does not mean the reference is wrong. Verify it manually.
436
+ - `DOI_NOT_IN_CROSSREF`: the DOI is registered with another agency (DataCite for arXiv and Zenodo, KISTI, JaLC, ...), so its metadata was not compared (`status: UNVERIFIED`, `verdict: WARN`). Open the DOI to confirm it.
437
+ - `DOI_NOT_FOUND`: neither CrossRef nor doi.org knows the DOI (`REJECT`).
438
+ - `PAPER_RETRACTED`, `ROW_CHECK_ERROR`: as for `check-claim` and `check-file`. Other DOI-backed results carry `error_code: null`; read `verdict`, `mismatches`, and `reason`, which names what differs (for example `the year differs (reference: 2009; CrossRef: 2010)`). A plain-text reference whose DOI belongs to a paper it does not mention is `status: MISMATCH`, `verdict: WARN`, with that paper's title in `reason`.
439
+
440
+ To hand the result to a co-author or supervisor, add `--report` to `check-bib`
441
+ or `check-file`. The file extension picks the format:
442
+
443
+ ```bash
444
+ ref-verify check-bib references.bib --report report.html
445
+ ref-verify check-file claims.jsonl --report report.md
446
+ ```
447
+
448
+ The HTML file is self-contained (inline CSS, no scripts, no external resources
449
+ other than `https://doi.org/` links). It opens with counts that add up to the
450
+ total (one box per verdict as shown), a plain-language line on what each
451
+ verdict means and asks you to do, then a "Needs a look" table with every
452
+ non-passing reference or claim and a "Passed" table below it. Each row is
453
+ coloured (`PASS`/`ACCEPT` green, `WARN` amber, `REJECT` red, `UNVERIFIED`
454
+ grey) and shows the reason and evidence. `UNVERIFIED` marks a result the tool
455
+ could not confirm automatically; it is not a finding that the reference is
456
+ wrong. The Markdown file has the same content. A `--report` path whose folder
457
+ does not exist is rejected before any lookup, so a long run is never lost.
458
+ `--json` output is unchanged.
459
+
279
460
  > Core rule: every content statement about a paper must come from a live-fetched
280
461
  > source at the depth the claim requires — abstract for topline claims, full
281
462
  > text for mechanism, implementation, or procedural claims. If the required
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "ref-verify"
7
- version = "1.2.2"
7
+ version = "1.3.1"
8
8
  description = "Executable DOI and claim verification helpers for academic citations"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -2,4 +2,4 @@
2
2
 
3
3
  __all__ = ["__version__"]
4
4
 
5
- __version__ = "1.2.2"
5
+ __version__ = "1.3.1"
@@ -0,0 +1,90 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ import os
6
+ import tempfile
7
+ import time
8
+ from dataclasses import dataclass
9
+ from pathlib import Path
10
+ from typing import Callable
11
+
12
+ DEFAULT_TTL_DAYS = 7.0
13
+ NOT_FOUND_TTL_SECONDS = 86400.0
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class CachedResponse:
18
+ status: int
19
+ body: str
20
+
21
+
22
+ class ResponseCache:
23
+ def __init__(
24
+ self,
25
+ directory: Path,
26
+ ttl_seconds: float,
27
+ *,
28
+ not_found_ttl_seconds: float = NOT_FOUND_TTL_SECONDS,
29
+ clock: Callable[[], float] = time.time,
30
+ ) -> None:
31
+ self.directory = directory
32
+ self.ttl_seconds = ttl_seconds
33
+ self.not_found_ttl_seconds = not_found_ttl_seconds
34
+ self.clock = clock
35
+
36
+ def get(self, url: str) -> CachedResponse | None:
37
+ try:
38
+ entry = json.loads(self._path(url).read_text(encoding="utf-8"))
39
+ status = entry["status"]
40
+ body = entry["body"]
41
+ stored_at = float(entry["stored_at"])
42
+ if entry["url"] != url or not isinstance(status, int) or not isinstance(body, str):
43
+ return None
44
+ except (OSError, ValueError, KeyError, TypeError):
45
+ return None
46
+ # A dead DOI gets a short TTL so a newly registered DOI is re-checked soon.
47
+ ttl = self.not_found_ttl_seconds if status == 404 else self.ttl_seconds
48
+ if self.clock() - stored_at >= ttl:
49
+ return None
50
+ return CachedResponse(status=status, body=body)
51
+
52
+ def put(self, url: str, status: int, body: str) -> None:
53
+ entry = {"url": url, "status": status, "stored_at": self.clock(), "body": body}
54
+ try:
55
+ self.directory.mkdir(parents=True, exist_ok=True)
56
+ # Write-then-rename so a parallel reader never sees a half-written entry.
57
+ with tempfile.NamedTemporaryFile(
58
+ "w",
59
+ encoding="utf-8",
60
+ dir=self.directory,
61
+ suffix=".tmp",
62
+ delete=False,
63
+ ) as handle:
64
+ json.dump(entry, handle)
65
+ os.replace(handle.name, self._path(url))
66
+ except OSError:
67
+ return
68
+
69
+ def _path(self, url: str) -> Path:
70
+ return self.directory / f"{hashlib.sha256(url.encode('utf-8')).hexdigest()}.json"
71
+
72
+
73
+ def cache_directory() -> Path:
74
+ explicit = os.environ.get("REF_VERIFY_CACHE_DIR")
75
+ if explicit:
76
+ return Path(explicit).expanduser()
77
+ xdg = os.environ.get("XDG_CACHE_HOME")
78
+ if xdg:
79
+ return Path(xdg).expanduser() / "ref-verify"
80
+ return Path.home() / ".cache" / "ref-verify"
81
+
82
+
83
+ def default_cache() -> ResponseCache | None:
84
+ if os.environ.get("REF_VERIFY_NO_CACHE", "").strip().lower() in ("1", "true", "yes"):
85
+ return None
86
+ try:
87
+ ttl_days = float(os.environ.get("REF_VERIFY_CACHE_TTL_DAYS", DEFAULT_TTL_DAYS))
88
+ except ValueError:
89
+ ttl_days = DEFAULT_TTL_DAYS
90
+ return ResponseCache(cache_directory(), ttl_seconds=ttl_days * 86400)