citesure 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- citesure/__init__.py +70 -0
- citesure/citations.py +321 -0
- citesure/cli.py +208 -0
- citesure/fetcher.py +527 -0
- citesure/mcp_server.py +213 -0
- citesure/models.py +147 -0
- citesure/nli.py +624 -0
- citesure/overlap.py +911 -0
- citesure/reachability.py +125 -0
- citesure/report.py +128 -0
- citesure-0.2.0.dist-info/METADATA +352 -0
- citesure-0.2.0.dist-info/RECORD +15 -0
- citesure-0.2.0.dist-info/WHEEL +5 -0
- citesure-0.2.0.dist-info/entry_points.txt +3 -0
- citesure-0.2.0.dist-info/top_level.txt +1 -0
citesure/__init__.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""citesure — verify that an LLM's claims are supported by the sources it cites.
|
|
2
|
+
|
|
3
|
+
Public API (library form, D5). The CLI and MCP server are thin wrappers over
|
|
4
|
+
these functions so ~80% of the logic is testable without any MCP plumbing.
|
|
5
|
+
|
|
6
|
+
Typical library usage::
|
|
7
|
+
|
|
8
|
+
from citesure import extract_citations, verify_citations, Report
|
|
9
|
+
|
|
10
|
+
citations = extract_citations("The sky is blue [1].", {"1": "https://..."})
|
|
11
|
+
report = verify_citations(citations) # tiers 1+2 by default (Phase 2)
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from .citations import Citation, extract_citations, load_input
|
|
17
|
+
from .models import Report, Status, Verdict
|
|
18
|
+
from .nli import (
|
|
19
|
+
DEFAULT_NLI_MODEL,
|
|
20
|
+
NLIError,
|
|
21
|
+
NLICrossEncoder,
|
|
22
|
+
apply_nli_tier,
|
|
23
|
+
get_nli_model,
|
|
24
|
+
resolve_nli_model,
|
|
25
|
+
score_nli,
|
|
26
|
+
score_nli_batch,
|
|
27
|
+
status_for_nli,
|
|
28
|
+
)
|
|
29
|
+
from .overlap import (
|
|
30
|
+
DEFAULT_TOP_K,
|
|
31
|
+
HIGH_OVERLAP_THRESHOLD,
|
|
32
|
+
LOW_OVERLAP_THRESHOLD,
|
|
33
|
+
rank_passages,
|
|
34
|
+
score_overlap,
|
|
35
|
+
segment_passages,
|
|
36
|
+
status_for_score,
|
|
37
|
+
)
|
|
38
|
+
from .reachability import classify_reachability, verify_citations
|
|
39
|
+
|
|
40
|
+
__version__ = "0.2.0"
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"__version__",
|
|
44
|
+
"Citation",
|
|
45
|
+
"Status",
|
|
46
|
+
"Verdict",
|
|
47
|
+
"Report",
|
|
48
|
+
"extract_citations",
|
|
49
|
+
"load_input",
|
|
50
|
+
"classify_reachability",
|
|
51
|
+
"verify_citations",
|
|
52
|
+
# overlap tier (Phase 2)
|
|
53
|
+
"score_overlap",
|
|
54
|
+
"rank_passages",
|
|
55
|
+
"segment_passages",
|
|
56
|
+
"status_for_score",
|
|
57
|
+
"DEFAULT_TOP_K",
|
|
58
|
+
"HIGH_OVERLAP_THRESHOLD",
|
|
59
|
+
"LOW_OVERLAP_THRESHOLD",
|
|
60
|
+
# NLI tier (Phase 3)
|
|
61
|
+
"NLIError",
|
|
62
|
+
"NLICrossEncoder",
|
|
63
|
+
"get_nli_model",
|
|
64
|
+
"resolve_nli_model",
|
|
65
|
+
"score_nli",
|
|
66
|
+
"score_nli_batch",
|
|
67
|
+
"status_for_nli",
|
|
68
|
+
"apply_nli_tier",
|
|
69
|
+
"DEFAULT_NLI_MODEL",
|
|
70
|
+
]
|
citesure/citations.py
ADDED
|
@@ -0,0 +1,321 @@
|
|
|
1
|
+
"""Citation extraction (D1).
|
|
2
|
+
|
|
3
|
+
Supported input forms:
|
|
4
|
+
|
|
5
|
+
* **Markdown, numeric markers** — ``The sky is blue [1].`` where ``1``
|
|
6
|
+
resolves through a URL map. The map comes either from an explicit
|
|
7
|
+
``url_map`` argument or from a trailing ``## Sources`` / ``## References``
|
|
8
|
+
section in the document (numbered list lines such as ``1. https://...`` or
|
|
9
|
+
reference-style definitions such as ``[1]: https://...``).
|
|
10
|
+
* **Markdown, inline links** — ``See [the docs](https://docs.example/api)``
|
|
11
|
+
where the URL lives in the link itself. The link label becomes the
|
|
12
|
+
``citation_id``.
|
|
13
|
+
* **Structured JSON** — a list of ``{"claim": ..., "citation": url-or-id}``
|
|
14
|
+
objects, optionally wrapped in ``{"citations": [...]}``. When
|
|
15
|
+
``citation`` is an id rather than a URL, it is resolved through an
|
|
16
|
+
optional top-level ``"sources"`` (or ``"url_map"``) dict. An optional
|
|
17
|
+
per-item ``"excerpt"`` field (D4) carries a user-supplied passage that
|
|
18
|
+
the overlap tier matches against instead of the fetched page.
|
|
19
|
+
|
|
20
|
+
Claim unit (D4): the *sentence* containing the marker is the claim text. If
|
|
21
|
+
the surrounding paragraph has no sentence boundary (no ``.``/``!``/``?``
|
|
22
|
+
followed by whitespace), the whole paragraph is the claim.
|
|
23
|
+
|
|
24
|
+
Missing-id policy: a numeric marker whose id is absent from the URL map is
|
|
25
|
+
**skipped** — no :class:`Citation` is emitted for it. ``extract_citations``
|
|
26
|
+
is a pure function with no side channel for warnings, so callers can detect
|
|
27
|
+
the gap by comparing the number of markers in the text to the number of
|
|
28
|
+
returned citations.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import json
|
|
34
|
+
import re
|
|
35
|
+
from dataclasses import dataclass, replace
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
from typing import Any
|
|
38
|
+
|
|
39
|
+
#: ``[1]`` style numeric marker, but NOT ``[1](url)`` (an inline link).
|
|
40
|
+
_NUMERIC_MARKER_RE = re.compile(r"\[(\d+)\](?!\()")
|
|
41
|
+
|
|
42
|
+
#: ``[label](target)`` inline link. Target: no spaces, no nested parens.
|
|
43
|
+
_INLINE_LINK_RE = re.compile(r"\[([^\[\]\n]+)\]\(\s*([^()\s]+)\s*\)")
|
|
44
|
+
|
|
45
|
+
#: A ``## Sources`` / ``## References`` heading (any level, up to 3 leading spaces).
|
|
46
|
+
_SOURCES_HEADING_RE = re.compile(
|
|
47
|
+
r"^\s{0,3}(#{1,6})\s*(sources|references)\b.*$", re.IGNORECASE | re.MULTILINE
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
#: Reference-style definition: ``[1]: https://example.com`` (optional title tail).
|
|
51
|
+
_REF_DEF_RE = re.compile(r"^\s*\[([^\]]+)\]:\s*<?(\S+?)>?(?:\s+.*)?$", re.MULTILINE)
|
|
52
|
+
|
|
53
|
+
#: Numbered list item: ``1. https://...`` or ``2) [Title](https://...)``.
|
|
54
|
+
_NUMBERED_ITEM_RE = re.compile(r"^\s*(\d+)[.)]\s+(\S.*)$", re.MULTILINE)
|
|
55
|
+
|
|
56
|
+
#: An explicit URL token inside a sources line.
|
|
57
|
+
_URL_TOKEN_RE = re.compile(r"(?:https?://|file://)[^\s\)\]>\"']+")
|
|
58
|
+
|
|
59
|
+
#: Paragraph separator: blank line(s).
|
|
60
|
+
_PARAGRAPH_SPLIT_RE = re.compile(r"\n\s*\n")
|
|
61
|
+
|
|
62
|
+
#: Sentence boundary: sentence-ending punctuation followed by whitespace.
|
|
63
|
+
_SENTENCE_BOUNDARY_RE = re.compile(r"(?<=[.!?])\s+")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass
|
|
67
|
+
class Citation:
|
|
68
|
+
"""One claim-to-source pairing extracted from an input document."""
|
|
69
|
+
|
|
70
|
+
citation_id: str
|
|
71
|
+
url: str
|
|
72
|
+
claim: str
|
|
73
|
+
#: Optional raw context around the citation in the source document.
|
|
74
|
+
source_text: str | None = None
|
|
75
|
+
#: Optional user-supplied excerpt (D4): when set, the overlap tier
|
|
76
|
+
#: matches the claim against this text instead of the fetched page.
|
|
77
|
+
#: Only the structured JSON input form carries it (``"excerpt"`` field).
|
|
78
|
+
excerpt: str | None = None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ---------------------------------------------------------------------------
|
|
82
|
+
# Sources-section parsing
|
|
83
|
+
# ---------------------------------------------------------------------------
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _split_sources_section(text: str) -> tuple[str, dict[str, str]]:
|
|
87
|
+
"""Split a markdown document into (body, url_map).
|
|
88
|
+
|
|
89
|
+
Everything from a ``## Sources``/``## References`` heading to the next
|
|
90
|
+
heading of the same or higher level (or EOF) is treated as the sources
|
|
91
|
+
block and removed from the body.
|
|
92
|
+
"""
|
|
93
|
+
m = _SOURCES_HEADING_RE.search(text)
|
|
94
|
+
if not m:
|
|
95
|
+
return text, {}
|
|
96
|
+
level = len(m.group(1))
|
|
97
|
+
body = text[: m.start()].rstrip()
|
|
98
|
+
rest = text[m.end() :]
|
|
99
|
+
stop = re.search(rf"^\s{{0,3}}#{{{1},{level}}}\s+\S", rest, re.MULTILINE)
|
|
100
|
+
block = rest[: stop.start()] if stop else rest
|
|
101
|
+
return body, _parse_sources_block(block)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _parse_sources_block(block: str) -> dict[str, str]:
|
|
105
|
+
"""Parse a sources block into ``{id: url}``.
|
|
106
|
+
|
|
107
|
+
Handles both reference-style definitions (``[1]: url``) and numbered
|
|
108
|
+
list lines (``1. url`` / ``2) [Title](url)`` / ``3. Some words url``).
|
|
109
|
+
"""
|
|
110
|
+
url_map: dict[str, str] = {}
|
|
111
|
+
for m in _REF_DEF_RE.finditer(block):
|
|
112
|
+
url_map[m.group(1).strip()] = m.group(2).strip()
|
|
113
|
+
for m in _NUMBERED_ITEM_RE.finditer(block):
|
|
114
|
+
num, rest = m.group(1), m.group(2).strip()
|
|
115
|
+
url = _extract_url_from_line(rest)
|
|
116
|
+
if url:
|
|
117
|
+
url_map.setdefault(num, url)
|
|
118
|
+
return url_map
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _extract_url_from_line(line: str) -> str | None:
|
|
122
|
+
"""Pull a URL out of a sources-list line (link, URL token, or bare path)."""
|
|
123
|
+
link = re.search(r"\[[^\]]*\]\(\s*([^()\s]+)\s*\)", line)
|
|
124
|
+
if link:
|
|
125
|
+
return link.group(1)
|
|
126
|
+
tok = _URL_TOKEN_RE.search(line)
|
|
127
|
+
if tok:
|
|
128
|
+
return tok.group(0)
|
|
129
|
+
candidate = line.strip().rstrip(".")
|
|
130
|
+
if candidate and " " not in candidate and not candidate.startswith("#"):
|
|
131
|
+
return candidate
|
|
132
|
+
return None
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# ---------------------------------------------------------------------------
|
|
136
|
+
# Claim-unit selection (D4)
|
|
137
|
+
# ---------------------------------------------------------------------------
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _claim_unit(paragraph: str, pos: int) -> str:
|
|
141
|
+
"""Return the sentence containing character offset ``pos``.
|
|
142
|
+
|
|
143
|
+
If the paragraph has no sentence boundary, the whole paragraph is the
|
|
144
|
+
claim (D4: "the sentence, or paragraph, if no sentence boundary").
|
|
145
|
+
"""
|
|
146
|
+
spans: list[tuple[int, int]] = []
|
|
147
|
+
start = 0
|
|
148
|
+
for m in _SENTENCE_BOUNDARY_RE.finditer(paragraph):
|
|
149
|
+
spans.append((start, m.start()))
|
|
150
|
+
start = m.end()
|
|
151
|
+
spans.append((start, len(paragraph)))
|
|
152
|
+
for s, e in spans:
|
|
153
|
+
if s <= pos < e:
|
|
154
|
+
seg = paragraph[s:e].strip()
|
|
155
|
+
return seg or paragraph.strip()
|
|
156
|
+
return paragraph.strip()
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# ---------------------------------------------------------------------------
|
|
160
|
+
# Public API
|
|
161
|
+
# ---------------------------------------------------------------------------
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def extract_citations(
|
|
165
|
+
text: str, url_map: dict[str, str] | None = None
|
|
166
|
+
) -> list[Citation]:
|
|
167
|
+
"""Extract citations from markdown text.
|
|
168
|
+
|
|
169
|
+
Handles numeric ``[n]`` markers (resolved via ``url_map`` or a trailing
|
|
170
|
+
Sources/References section) and inline ``[label](url)`` links. The claim
|
|
171
|
+
for each marker is the sentence (or paragraph) containing it (D4).
|
|
172
|
+
|
|
173
|
+
Numeric markers whose id is missing from the URL map are skipped (see
|
|
174
|
+
module docstring for the rationale).
|
|
175
|
+
"""
|
|
176
|
+
body, section_map = _split_sources_section(text)
|
|
177
|
+
merged: dict[str, str] = dict(section_map)
|
|
178
|
+
if url_map:
|
|
179
|
+
merged.update({str(k): v for k, v in url_map.items()})
|
|
180
|
+
|
|
181
|
+
citations: list[Citation] = []
|
|
182
|
+
for para in _PARAGRAPH_SPLIT_RE.split(body):
|
|
183
|
+
para = para.strip()
|
|
184
|
+
if not para:
|
|
185
|
+
continue
|
|
186
|
+
events: list[tuple[int, str, str, str]] = [] # (pos, kind, payload, label)
|
|
187
|
+
for m in _NUMERIC_MARKER_RE.finditer(para):
|
|
188
|
+
events.append((m.start(), "num", m.group(1), m.group(1)))
|
|
189
|
+
for m in _INLINE_LINK_RE.finditer(para):
|
|
190
|
+
target = m.group(2)
|
|
191
|
+
if target.startswith("#") or target.lower().startswith("mailto:"):
|
|
192
|
+
continue
|
|
193
|
+
events.append((m.start(), "link", target, m.group(1).strip()))
|
|
194
|
+
events.sort(key=lambda e: e[0])
|
|
195
|
+
for pos, kind, payload, label in events:
|
|
196
|
+
if kind == "num":
|
|
197
|
+
url = merged.get(payload)
|
|
198
|
+
if url is None:
|
|
199
|
+
continue # missing id in url_map → skip (documented)
|
|
200
|
+
citations.append(
|
|
201
|
+
Citation(citation_id=payload, url=url, claim=_claim_unit(para, pos))
|
|
202
|
+
)
|
|
203
|
+
else:
|
|
204
|
+
citations.append(
|
|
205
|
+
Citation(citation_id=label, url=payload, claim=_claim_unit(para, pos))
|
|
206
|
+
)
|
|
207
|
+
return citations
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
# ---------------------------------------------------------------------------
|
|
211
|
+
# JSON input
|
|
212
|
+
# ---------------------------------------------------------------------------
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _looks_like_path(value: str) -> bool:
|
|
216
|
+
"""True for URLs (any scheme) or filesystem paths (absolute, relative
|
|
217
|
+
with a slash, or ``~``)."""
|
|
218
|
+
v = value.strip()
|
|
219
|
+
if "://" in v:
|
|
220
|
+
return True
|
|
221
|
+
return v.startswith(("/", "./", "../", "~")) or "/" in v
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _citations_from_json(data: Any) -> list[Citation]:
|
|
225
|
+
"""Parse structured JSON input into citations (D1)."""
|
|
226
|
+
if isinstance(data, dict):
|
|
227
|
+
if "citations" not in data:
|
|
228
|
+
raise ValueError(
|
|
229
|
+
"JSON input must be a list of {claim, citation} objects or an "
|
|
230
|
+
"object with a 'citations' key"
|
|
231
|
+
)
|
|
232
|
+
items = data["citations"]
|
|
233
|
+
sources = data.get("sources") or data.get("url_map") or {}
|
|
234
|
+
if not isinstance(sources, dict):
|
|
235
|
+
raise ValueError("'sources' must be a mapping of id -> url")
|
|
236
|
+
elif isinstance(data, list):
|
|
237
|
+
items, sources = data, {}
|
|
238
|
+
else:
|
|
239
|
+
raise ValueError("JSON input must be a list or an object")
|
|
240
|
+
|
|
241
|
+
out: list[Citation] = []
|
|
242
|
+
if items is None:
|
|
243
|
+
return out
|
|
244
|
+
for i, item in enumerate(items, 1):
|
|
245
|
+
if not isinstance(item, dict) or "claim" not in item or "citation" not in item:
|
|
246
|
+
raise ValueError(
|
|
247
|
+
f"citation #{i}: expected an object with 'claim' and 'citation' keys"
|
|
248
|
+
)
|
|
249
|
+
cid = str(item.get("id") or f"c{i}")
|
|
250
|
+
ref = str(item["citation"]).strip()
|
|
251
|
+
if "{{" in ref or "}}" in ref:
|
|
252
|
+
raise ValueError(
|
|
253
|
+
f"citation #{i} ({cid}): citation '{ref}' contains an "
|
|
254
|
+
"unresolved template placeholder (e.g. {{MOCK}}) — substitute "
|
|
255
|
+
"a concrete URL before verifying"
|
|
256
|
+
)
|
|
257
|
+
if _looks_like_path(ref):
|
|
258
|
+
url = ref
|
|
259
|
+
elif ref in sources:
|
|
260
|
+
url = sources[ref]
|
|
261
|
+
else:
|
|
262
|
+
raise ValueError(
|
|
263
|
+
f"citation #{i} ({cid}): citation '{ref}' is not a URL, not a "
|
|
264
|
+
"path, and there is no matching entry in the 'sources' map"
|
|
265
|
+
)
|
|
266
|
+
out.append(
|
|
267
|
+
Citation(
|
|
268
|
+
citation_id=cid,
|
|
269
|
+
url=url,
|
|
270
|
+
claim=str(item["claim"]),
|
|
271
|
+
source_text=item.get("source_text"),
|
|
272
|
+
excerpt=(str(item["excerpt"]).strip() or None)
|
|
273
|
+
if item.get("excerpt") is not None
|
|
274
|
+
else None,
|
|
275
|
+
)
|
|
276
|
+
)
|
|
277
|
+
return out
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
# ---------------------------------------------------------------------------
|
|
281
|
+
# File loading
|
|
282
|
+
# ---------------------------------------------------------------------------
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _anchor(citation: Citation, base: Path) -> Citation:
|
|
286
|
+
"""Resolve a bare relative path in ``citation.url`` against ``base``.
|
|
287
|
+
|
|
288
|
+
Scheme URLs (http/https/file) and absolute paths are left untouched.
|
|
289
|
+
Anchoring happens in :func:`load_input` so that bundled samples work
|
|
290
|
+
regardless of the current working directory.
|
|
291
|
+
"""
|
|
292
|
+
u = citation.url
|
|
293
|
+
if "://" in u or u.startswith(("/", "~")):
|
|
294
|
+
return citation
|
|
295
|
+
return replace(citation, url=str((base / u).resolve()))
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def load_input(path: str) -> tuple[list[Citation], dict[str, Any]]:
|
|
299
|
+
"""Read a file and auto-detect its format (D1).
|
|
300
|
+
|
|
301
|
+
* Structured JSON (list of ``{claim, citation}`` pairs, optionally
|
|
302
|
+
wrapped in ``{"citations": [...]}``) → ``meta["format"] == "json"``.
|
|
303
|
+
* Anything else is treated as markdown with inline citations →
|
|
304
|
+
``meta["format"] == "markdown"``.
|
|
305
|
+
|
|
306
|
+
Returns ``(citations, meta)``. Bare relative paths in citation URLs are
|
|
307
|
+
resolved against the input file's directory.
|
|
308
|
+
"""
|
|
309
|
+
p = Path(path)
|
|
310
|
+
raw = p.read_text(encoding="utf-8")
|
|
311
|
+
stripped = raw.lstrip()
|
|
312
|
+
if stripped[:1] in ("{", "["):
|
|
313
|
+
try:
|
|
314
|
+
data = json.loads(raw)
|
|
315
|
+
except json.JSONDecodeError:
|
|
316
|
+
data = None
|
|
317
|
+
if data is not None:
|
|
318
|
+
citations = _citations_from_json(data)
|
|
319
|
+
return [_anchor(c, p.parent) for c in citations], {"format": "json"}
|
|
320
|
+
citations = extract_citations(raw)
|
|
321
|
+
return [_anchor(c, p.parent) for c in citations], {"format": "markdown"}
|
citesure/cli.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""Command-line interface: ``citesure verify <file>`` (D5/D12).
|
|
2
|
+
|
|
3
|
+
Usage::
|
|
4
|
+
|
|
5
|
+
citesure verify tests/fixtures/sample.md # human-readable
|
|
6
|
+
citesure verify tests/fixtures/sample.md --json # machine-readable
|
|
7
|
+
citesure verify sample.json --threshold 0.9 --strict
|
|
8
|
+
citesure verify sample.md --md report.md # also write a .md file
|
|
9
|
+
citesure verify sample.md --nli # + NLI entailment tier
|
|
10
|
+
citesure verify sample.md --nli --nli-model NAME # custom cross-encoder
|
|
11
|
+
|
|
12
|
+
Exit codes (D3):
|
|
13
|
+
|
|
14
|
+
* ``0`` — default mode: iff ``pass_rate >= --threshold`` (default 0.8).
|
|
15
|
+
With ``--strict`` (CI mode): additionally requires **no** ``ambiguous``
|
|
16
|
+
AND **no** ``unsupported`` verdicts — ambiguous counts as an explicit
|
|
17
|
+
failure.
|
|
18
|
+
* ``1`` — verification ran but the exit rule above is not met.
|
|
19
|
+
* ``2`` — input error (unreadable file, malformed JSON, no citations found
|
|
20
|
+
in a JSON input that must have some) OR the NLI model failed to load
|
|
21
|
+
(fail fast, D6 — distinct from a verification-failure exit 1).
|
|
22
|
+
|
|
23
|
+
The default pipeline runs both D2 tiers: reachability (tier 1) and content
|
|
24
|
+
overlap (tier 2); ``tier_reached`` in each verdict records how far it got.
|
|
25
|
+
|
|
26
|
+
``--nli`` enables the NLI entailment tier (tier 3, D2/D6): a local
|
|
27
|
+
cross-encoder scores each (claim, best-passage) pair and the D3 banding
|
|
28
|
+
produces the final status. ``--nli-model NAME`` selects the model via the
|
|
29
|
+
D6 priority chain (flag → ``CITECHECK_NLI_MODEL`` env → built-in default
|
|
30
|
+
``cross-encoder/nli-deberta-v3-base``). The model is lazy-downloaded on
|
|
31
|
+
first use (~425 MB) into ``~/.cache/citesure/`` (override:
|
|
32
|
+
``CITECHECK_CACHE_DIR``). When NLI is on, the report header shows which
|
|
33
|
+
model was used.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import argparse
|
|
39
|
+
import asyncio
|
|
40
|
+
import json
|
|
41
|
+
import os
|
|
42
|
+
import sys
|
|
43
|
+
from pathlib import Path
|
|
44
|
+
|
|
45
|
+
from .citations import load_input
|
|
46
|
+
from .models import Report
|
|
47
|
+
from .nli import NLIError, get_nli_model, resolve_nli_model
|
|
48
|
+
from .reachability import verify_citations
|
|
49
|
+
from .report import exit_code, render_human, render_markdown
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
53
|
+
parser = argparse.ArgumentParser(
|
|
54
|
+
prog="citesure",
|
|
55
|
+
description=(
|
|
56
|
+
"Verify that an LLM's claims are supported by the sources it cites."
|
|
57
|
+
),
|
|
58
|
+
)
|
|
59
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
60
|
+
|
|
61
|
+
v = sub.add_parser(
|
|
62
|
+
"verify", help="verify citations in a markdown or JSON file"
|
|
63
|
+
)
|
|
64
|
+
v.add_argument(
|
|
65
|
+
"file",
|
|
66
|
+
help="input file: markdown with inline citations, or structured JSON",
|
|
67
|
+
)
|
|
68
|
+
v.add_argument(
|
|
69
|
+
"--json",
|
|
70
|
+
action="store_true",
|
|
71
|
+
help="print the full report as JSON instead of human-readable text",
|
|
72
|
+
)
|
|
73
|
+
v.add_argument(
|
|
74
|
+
"--md",
|
|
75
|
+
metavar="FILE",
|
|
76
|
+
help="also write the Markdown report to FILE (in addition to stdout)",
|
|
77
|
+
)
|
|
78
|
+
v.add_argument(
|
|
79
|
+
"--threshold",
|
|
80
|
+
type=float,
|
|
81
|
+
default=0.8,
|
|
82
|
+
help=(
|
|
83
|
+
"minimum pass_rate for exit code 0 (default: 0.8). Exit code is "
|
|
84
|
+
"0 iff pass_rate >= threshold; with --strict, ambiguous and "
|
|
85
|
+
"unsupported verdicts are additionally treated as failures."
|
|
86
|
+
),
|
|
87
|
+
)
|
|
88
|
+
v.add_argument(
|
|
89
|
+
"--strict",
|
|
90
|
+
action="store_true",
|
|
91
|
+
help=(
|
|
92
|
+
"CI mode: ambiguous verdicts count as explicit failures. Exit "
|
|
93
|
+
"code is 0 only if pass_rate >= --threshold AND there are no "
|
|
94
|
+
"ambiguous AND no unsupported verdicts; otherwise 1."
|
|
95
|
+
),
|
|
96
|
+
)
|
|
97
|
+
v.add_argument(
|
|
98
|
+
"--nli",
|
|
99
|
+
action="store_true",
|
|
100
|
+
default=True,
|
|
101
|
+
help=(
|
|
102
|
+
"enable the NLI entailment tier (tier 3): a local cross-encoder "
|
|
103
|
+
"scores each (claim, best-passage) pair. Lazy-downloads the "
|
|
104
|
+
"default model (~425 MB) into ~/.cache/citesure/ on first use. "
|
|
105
|
+
"(default: on; use --no-nli to disable)"
|
|
106
|
+
),
|
|
107
|
+
)
|
|
108
|
+
v.add_argument(
|
|
109
|
+
"--no-nli",
|
|
110
|
+
action="store_false",
|
|
111
|
+
dest="nli",
|
|
112
|
+
help=(
|
|
113
|
+
"disable the NLI entailment tier; overlap-only verdicts cannot "
|
|
114
|
+
"detect negation or entity-swap claims."
|
|
115
|
+
),
|
|
116
|
+
)
|
|
117
|
+
v.add_argument(
|
|
118
|
+
"--nli-model",
|
|
119
|
+
metavar="NAME",
|
|
120
|
+
help=(
|
|
121
|
+
"cross-encoder model name (Hugging Face id) or local path; "
|
|
122
|
+
"highest priority in the D6 chain (flag > CITECHECK_NLI_MODEL "
|
|
123
|
+
"env > built-in default). Implies --nli."
|
|
124
|
+
),
|
|
125
|
+
)
|
|
126
|
+
v.add_argument(
|
|
127
|
+
"--cache-dir",
|
|
128
|
+
metavar="DIR",
|
|
129
|
+
help="override the fetch cache directory (sets CITECHECK_CACHE_DIR)",
|
|
130
|
+
)
|
|
131
|
+
return parser
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# ---------------------------------------------------------------------------
|
|
135
|
+
# Command implementation
|
|
136
|
+
# ---------------------------------------------------------------------------
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _cmd_verify(args: argparse.Namespace) -> int:
|
|
140
|
+
if args.cache_dir:
|
|
141
|
+
os.environ["CITECHECK_CACHE_DIR"] = str(
|
|
142
|
+
Path(args.cache_dir).expanduser().resolve()
|
|
143
|
+
)
|
|
144
|
+
# --nli-model implies --nli (selecting a model means using the tier).
|
|
145
|
+
use_nli = bool(args.nli or args.nli_model)
|
|
146
|
+
if not use_nli:
|
|
147
|
+
print(
|
|
148
|
+
"citesure: WARNING --no-nli disables the entailment tier; "
|
|
149
|
+
"overlap-only verdicts cannot detect negation or entity-swap claims "
|
|
150
|
+
"(see evals/EVAL_REPORT.md).",
|
|
151
|
+
file=sys.stderr,
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
try:
|
|
155
|
+
citations, meta = load_input(args.file)
|
|
156
|
+
except (OSError, ValueError) as exc:
|
|
157
|
+
print(f"citesure: cannot read input: {exc}", file=sys.stderr)
|
|
158
|
+
return 2
|
|
159
|
+
if not citations:
|
|
160
|
+
print("citesure: warning: no citations found in input", file=sys.stderr)
|
|
161
|
+
|
|
162
|
+
nli_name: str | None = None
|
|
163
|
+
if use_nli:
|
|
164
|
+
# Fail fast BEFORE any fetching/scoring (D6): a bad model name must
|
|
165
|
+
# not burn a full verification run only to die at the end.
|
|
166
|
+
try:
|
|
167
|
+
get_nli_model(args.nli_model)
|
|
168
|
+
except NLIError as exc:
|
|
169
|
+
print(f"citesure: {exc}", file=sys.stderr)
|
|
170
|
+
return 2
|
|
171
|
+
nli_name = resolve_nli_model(args.nli_model)
|
|
172
|
+
|
|
173
|
+
# Pipeline: tier 1 (reachability) + tier 2 (content overlap), plus tier 3
|
|
174
|
+
# (NLI entailment) when enabled.
|
|
175
|
+
try:
|
|
176
|
+
report = asyncio.run(
|
|
177
|
+
verify_citations(
|
|
178
|
+
citations, use_overlap=True, use_nli=use_nli, nli_model=args.nli_model
|
|
179
|
+
)
|
|
180
|
+
)
|
|
181
|
+
except NLIError as exc: # defensive: load already validated above
|
|
182
|
+
print(f"citesure: {exc}", file=sys.stderr)
|
|
183
|
+
return 2
|
|
184
|
+
|
|
185
|
+
if args.json:
|
|
186
|
+
print(json.dumps(report.to_dict(), indent=2))
|
|
187
|
+
else:
|
|
188
|
+
print(render_human(report, meta, args.threshold, args.strict, nli_model=nli_name))
|
|
189
|
+
if args.md:
|
|
190
|
+
Path(args.md).write_text(
|
|
191
|
+
render_markdown(
|
|
192
|
+
report, meta, args.threshold, args.strict, nli_model=nli_name
|
|
193
|
+
),
|
|
194
|
+
encoding="utf-8",
|
|
195
|
+
)
|
|
196
|
+
return exit_code(report, args.threshold, args.strict)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def main(argv: list[str] | None = None) -> int:
|
|
200
|
+
"""Console-script entry point (``citesure``)."""
|
|
201
|
+
args = _build_parser().parse_args(argv)
|
|
202
|
+
if args.command == "verify":
|
|
203
|
+
return _cmd_verify(args)
|
|
204
|
+
return 2 # pragma: no cover - argparse enforces the subcommand
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
if __name__ == "__main__":
|
|
208
|
+
raise SystemExit(main())
|