hallucination-check 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ # Virtual environments and scratch space used while building. Never committed:
2
+ # they are large, machine-specific, and rebuilt from pyproject.toml anyway.
3
+ _venvs/
4
+ _proof/
5
+ .venv*/
6
+
7
+ # Build output. Wheels are built by the release workflow from this source, so a
8
+ # wheel in git could silently differ from the code beside it.
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ src/*.egg-info/
13
+
14
+ # Python noise
15
+ __pycache__/
16
+ *.py[cod]
17
+ .pytest_cache/
18
+ .mypy_cache/
19
+ .ruff_cache/
20
+
21
+ # Local verification state, not source
22
+ verify_results.json
23
+ publish-log.txt
24
+ published.json
25
+
26
+ # Credentials. None of these belong here, and this line is the backstop.
27
+ .pypirc
28
+ .env
29
+ *.pem
30
+ *.key
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Pranay Mahendrakar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,176 @@
1
+ Metadata-Version: 2.5
2
+ Name: hallucination-check
3
+ Version: 0.1.0
4
+ Summary: Check an answer against the sources it claims to use and flag every unsupported sentence
5
+ Project-URL: Homepage, https://pypi.org/project/hallucination-check/
6
+ Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
7
+ Author: Pranay Mahendrakar
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: attribution,citation,fact-checking,faithfulness,grounding,hallucination,llm-evaluation,nlp,rag
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Topic :: Text Processing :: Linguistic
19
+ Requires-Python: >=3.9
20
+ Requires-Dist: numpy>=1.23
21
+ Provides-Extra: dev
22
+ Requires-Dist: pytest>=7; extra == 'dev'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # hallucination-check
26
+
27
+ Check an answer against the sources it claims to use, and get back the exact sentences the
28
+ sources do not support.
29
+
30
+ ## Install
31
+
32
+ ```
33
+ pip install hallucination-check
34
+ ```
35
+
36
+ Only dependency: numpy. Nothing is downloaded at import or at run time.
37
+
38
+ ## Quickstart
39
+
40
+ ```python
41
+ import hallucination_check as hc
42
+
43
+ answer = "The Eiffel Tower is in Paris. It was completed in 1889. It cost 42 million francs."
44
+ sources = ["The Eiffel Tower stands in Paris, France, and was completed in 1889."]
45
+ report = hc.check(answer, sources)
46
+ print(report.summary()) # 66.7 / 100 grounded, 1 claim flagged
47
+ print([c.text for c in report.unsupported]) # ['It cost 42 million francs.']
48
+ ```
49
+
50
+ `sources` can be one string, a list of strings, or a list of dicts with `"text"` and an
51
+ optional `"id"`, so retrieved passages go straight in:
52
+
53
+ ```python
54
+ hc.check(answer, [{"id": "wiki:eiffel#3", "text": "..."}, {"id": "wiki:paris#1", "text": "..."}])
55
+ ```
56
+
57
+ ## What it checks
58
+
59
+ - The answer is split into claims (one sentence each by default; `granularity="clause"` or
60
+ `"paragraph"` also work). Every claim is checked on its own.
61
+ - For each claim the sources are searched for the best-matching span of one to three
62
+ consecutive sentences. That span, its source id and the score are on the claim, so every
63
+ verdict can be traced back to the text it came from.
64
+ - The support score blends three lexical signals: character 4-gram cosine, word
65
+ unigram/bigram cosine, and how much of the claim's content vocabulary the span covers
66
+ (light plural and tense stripping, so "site" matches "sites"). A claim scoring at or above
67
+ `threshold` (default 0.55) counts as supported.
68
+ - **A claim carrying a number, date, name or quantity that appears nowhere in the sources is
69
+ flagged whatever its similarity**, because those are what people actually get wrong.
70
+ `12%`, `$4.3 billion`, `1,200`, `twenty-three`, `12 March 2024`, `November 2022` and
71
+ `NASA` are all recognized; numbers written in words and in digits compare equal.
72
+ - A fact has to appear in the *same role*, not merely somewhere in the pile of passages.
73
+ A measurement is tied to the unit or head noun after it and a name to the words around
74
+ it, so "reduced symptoms by 1,200 percent" is not backed up by "enrolled 1,200
75
+ participants" in another passage, "11 metres" is not backed up by the 11 in "Apollo 11",
76
+ and "developed at Pfizer" is not backed up by a passage that only mentions Pfizer's
77
+ earnings. A possessive in the source counts as the bare name, so "France's capital"
78
+ grounds a claim about France.
79
+ - When the best span states a *different* number or date for otherwise the same sentence,
80
+ the claim also lands in `report.contradictions` with a reason like
81
+ `the sources say 1,200 where the answer says 4,200`.
82
+ - A capitalized word at the start of a sentence is not treated as a name unless it is an
83
+ acronym or the start of a multi-word name, so "Cats sleep a lot." is not flagged for
84
+ "Cats".
85
+ - Everything is deterministic: the same answer and sources always give the same report.
86
+ Nothing is written, nothing is fetched, and the library never prints.
87
+
88
+ **Lexical grounding is a signal, not proof.** A claim can score well while quietly reversing
89
+ the source's meaning, and a correct paraphrase that shares no vocabulary will score badly.
90
+ The `embed` hook is how you strengthen it: pass any function that turns a list of strings
91
+ into vectors, and semantic similarity is blended into the score at equal weight (this
92
+ package stays free of model dependencies, so the model stays yours).
93
+
94
+ ```python
95
+ from sentence_transformers import SentenceTransformer # your choice, not a dependency
96
+ model = SentenceTransformer("all-MiniLM-L6-v2")
97
+ report = hc.check(answer, sources, embed=lambda texts: model.encode(texts))
98
+ ```
99
+
100
+ With `embed` supplied, every source sentence is embedded (up to `embed_max_sentences`), so
101
+ the hook also finds spans that share no words at all with the claim.
102
+
103
+ ## API
104
+
105
+ ```python
106
+ hc.check(answer, sources, *, threshold=0.55, granularity="sentence", embed=None) -> GroundingReport
107
+ hc.check_batch(pairs, **kw) -> list[GroundingReport]
108
+ ```
109
+
110
+ `pairs` is an iterable of `(answer, sources)` tuples, or of dicts with `"answer"` and
111
+ `"sources"` keys. `**kw` goes to `GroundingChecker`.
112
+
113
+ `GroundingReport`
114
+
115
+ - `.score` - 0-100, the share of the answer's claims the sources support
116
+ - `.claims` - `list[Claim]` in answer order
117
+ - `.unsupported` - the claims the sources do not back up
118
+ - `.contradictions` - unsupported claims where a source states a different number or date
119
+ - `.citations` - `dict[claim index -> source id]` for the supported claims
120
+ - `.n_claims`, `.n_supported`, `.threshold`, `.granularity`, `.source_ids`, `.warnings`
121
+ - `.summary()` - human-readable text; `.to_dict()` - JSON-safe dict
122
+
123
+ `Claim(text, supported, confidence, best_source, best_span, reason)`
124
+
125
+ - `.text` - the claim as it appears in the answer
126
+ - `.supported` - the verdict
127
+ - `.confidence` - 0-1 support score for this claim
128
+ - `.best_source` - id of the source the best span came from (`"s1"`, `"s2"`, ... unless you
129
+ gave ids), or `None` when nothing matched
130
+ - `.best_span` - the source text that matched best
131
+ - `.reason` - one plain sentence saying why, e.g.
132
+ `these appear nowhere in the sources: 42 million`
133
+ - `.to_dict()`
134
+
135
+ ```python
136
+ report = hc.check(answer, sources)
137
+ report.score # 66.7
138
+ report.claims[1].best_span # 'The Eiffel Tower stands in Paris, France, and was completed in 1889.'
139
+ report.citations # {0: 's1', 1: 's1'}
140
+ report.to_dict()["unsupported"] # [2]
141
+ ```
142
+
143
+ `GroundingChecker(threshold=0.55, granularity="sentence", embed=None, embed_weight=0.5,
144
+ flag_missing_facts=True, char_n=4, window=3, top_k=25, embed_top=5,
145
+ embed_max_sentences=2000)` is the class underneath, with `.check(answer, sources)` and
146
+ `.check_batch(pairs)`. Set `flag_missing_facts=False` to score on similarity alone.
147
+
148
+ Edge cases are settled, not accidental: an empty answer scores 100 with no claims; empty
149
+ sources score 0 with every claim unsupported; an answer identical to a source scores 100;
150
+ the score is always between 0 and 100.
151
+
152
+ ## CLI
153
+
154
+ ```
155
+ hallucination-check ANSWER [-s SOURCE ...] [--threshold 0.55]
156
+ [--granularity sentence|clause|paragraph] [--no-fact-flags]
157
+ [--json] [--output PATH] [--fail-under SCORE]
158
+ ```
159
+
160
+ `ANSWER` and each `SOURCE` may be a file path or the text itself; `-` reads stdin. A value
161
+ that can only have meant a file -- one bare word ending in `.txt`, `.md`, `.json`, `.jsonl`
162
+ or `.ndjson` -- is an error when no such file exists, rather than being checked as literal
163
+ text, so a mistyped path never turns into a confident wrong report.
164
+
165
+ - `hallucination-check answer.txt -s passage1.txt -s passage2.txt` prints the summary.
166
+ - `hallucination-check "Paris is the capital of France." -s "France's capital is Paris."`
167
+ takes text directly.
168
+ - `cat answer.txt | hallucination-check - -s retrieved.jsonl --json` reads the answer from
169
+ stdin and the passages from a JSON-lines file (one object per line with `text` and an
170
+ optional `id`), and prints `to_dict()` as JSON.
171
+ - `--output PATH` writes that JSON to a file; `--fail-under 80` exits with status 2 when the
172
+ score is below 80, which is the useful thing to put in CI.
173
+
174
+ ## License
175
+
176
+ MIT
@@ -0,0 +1,152 @@
1
+ # hallucination-check
2
+
3
+ Check an answer against the sources it claims to use, and get back the exact sentences the
4
+ sources do not support.
5
+
6
+ ## Install
7
+
8
+ ```
9
+ pip install hallucination-check
10
+ ```
11
+
12
+ Only dependency: numpy. Nothing is downloaded at import or at run time.
13
+
14
+ ## Quickstart
15
+
16
+ ```python
17
+ import hallucination_check as hc
18
+
19
+ answer = "The Eiffel Tower is in Paris. It was completed in 1889. It cost 42 million francs."
20
+ sources = ["The Eiffel Tower stands in Paris, France, and was completed in 1889."]
21
+ report = hc.check(answer, sources)
22
+ print(report.summary()) # 66.7 / 100 grounded, 1 claim flagged
23
+ print([c.text for c in report.unsupported]) # ['It cost 42 million francs.']
24
+ ```
25
+
26
+ `sources` can be one string, a list of strings, or a list of dicts with `"text"` and an
27
+ optional `"id"`, so retrieved passages go straight in:
28
+
29
+ ```python
30
+ hc.check(answer, [{"id": "wiki:eiffel#3", "text": "..."}, {"id": "wiki:paris#1", "text": "..."}])
31
+ ```
32
+
33
+ ## What it checks
34
+
35
+ - The answer is split into claims (one sentence each by default; `granularity="clause"` or
36
+ `"paragraph"` also work). Every claim is checked on its own.
37
+ - For each claim the sources are searched for the best-matching span of one to three
38
+ consecutive sentences. That span, its source id and the score are on the claim, so every
39
+ verdict can be traced back to the text it came from.
40
+ - The support score blends three lexical signals: character 4-gram cosine, word
41
+ unigram/bigram cosine, and how much of the claim's content vocabulary the span covers
42
+ (light plural and tense stripping, so "site" matches "sites"). A claim scoring at or above
43
+ `threshold` (default 0.55) counts as supported.
44
+ - **A claim carrying a number, date, name or quantity that appears nowhere in the sources is
45
+ flagged whatever its similarity**, because those are what people actually get wrong.
46
+ `12%`, `$4.3 billion`, `1,200`, `twenty-three`, `12 March 2024`, `November 2022` and
47
+ `NASA` are all recognized; numbers written in words and in digits compare equal.
48
+ - A fact has to appear in the *same role*, not merely somewhere in the pile of passages.
49
+ A measurement is tied to the unit or head noun after it and a name to the words around
50
+ it, so "reduced symptoms by 1,200 percent" is not backed up by "enrolled 1,200
51
+ participants" in another passage, "11 metres" is not backed up by the 11 in "Apollo 11",
52
+ and "developed at Pfizer" is not backed up by a passage that only mentions Pfizer's
53
+ earnings. A possessive in the source counts as the bare name, so "France's capital"
54
+ grounds a claim about France.
55
+ - When the best span states a *different* number or date for otherwise the same sentence,
56
+ the claim also lands in `report.contradictions` with a reason like
57
+ `the sources say 1,200 where the answer says 4,200`.
58
+ - A capitalized word at the start of a sentence is not treated as a name unless it is an
59
+ acronym or the start of a multi-word name, so "Cats sleep a lot." is not flagged for
60
+ "Cats".
61
+ - Everything is deterministic: the same answer and sources always give the same report.
62
+ Nothing is written, nothing is fetched, and the library never prints.
63
+
64
+ **Lexical grounding is a signal, not proof.** A claim can score well while quietly reversing
65
+ the source's meaning, and a correct paraphrase that shares no vocabulary will score badly.
66
+ The `embed` hook is how you strengthen it: pass any function that turns a list of strings
67
+ into vectors, and semantic similarity is blended into the score at equal weight (this
68
+ package stays free of model dependencies, so the model stays yours).
69
+
70
+ ```python
71
+ from sentence_transformers import SentenceTransformer # your choice, not a dependency
72
+ model = SentenceTransformer("all-MiniLM-L6-v2")
73
+ report = hc.check(answer, sources, embed=lambda texts: model.encode(texts))
74
+ ```
75
+
76
+ With `embed` supplied, every source sentence is embedded (up to `embed_max_sentences`), so
77
+ the hook also finds spans that share no words at all with the claim.
78
+
79
+ ## API
80
+
81
+ ```python
82
+ hc.check(answer, sources, *, threshold=0.55, granularity="sentence", embed=None) -> GroundingReport
83
+ hc.check_batch(pairs, **kw) -> list[GroundingReport]
84
+ ```
85
+
86
+ `pairs` is an iterable of `(answer, sources)` tuples, or of dicts with `"answer"` and
87
+ `"sources"` keys. `**kw` goes to `GroundingChecker`.
88
+
89
+ `GroundingReport`
90
+
91
+ - `.score` - 0-100, the share of the answer's claims the sources support
92
+ - `.claims` - `list[Claim]` in answer order
93
+ - `.unsupported` - the claims the sources do not back up
94
+ - `.contradictions` - unsupported claims where a source states a different number or date
95
+ - `.citations` - `dict[claim index -> source id]` for the supported claims
96
+ - `.n_claims`, `.n_supported`, `.threshold`, `.granularity`, `.source_ids`, `.warnings`
97
+ - `.summary()` - human-readable text; `.to_dict()` - JSON-safe dict
98
+
99
+ `Claim(text, supported, confidence, best_source, best_span, reason)`
100
+
101
+ - `.text` - the claim as it appears in the answer
102
+ - `.supported` - the verdict
103
+ - `.confidence` - 0-1 support score for this claim
104
+ - `.best_source` - id of the source the best span came from (`"s1"`, `"s2"`, ... unless you
105
+ gave ids), or `None` when nothing matched
106
+ - `.best_span` - the source text that matched best
107
+ - `.reason` - one plain sentence saying why, e.g.
108
+ `these appear nowhere in the sources: 42 million`
109
+ - `.to_dict()`
110
+
111
+ ```python
112
+ report = hc.check(answer, sources)
113
+ report.score # 66.7
114
+ report.claims[1].best_span # 'The Eiffel Tower stands in Paris, France, and was completed in 1889.'
115
+ report.citations # {0: 's1', 1: 's1'}
116
+ report.to_dict()["unsupported"] # [2]
117
+ ```
118
+
119
+ `GroundingChecker(threshold=0.55, granularity="sentence", embed=None, embed_weight=0.5,
120
+ flag_missing_facts=True, char_n=4, window=3, top_k=25, embed_top=5,
121
+ embed_max_sentences=2000)` is the class underneath, with `.check(answer, sources)` and
122
+ `.check_batch(pairs)`. Set `flag_missing_facts=False` to score on similarity alone.
123
+
124
+ Edge cases are settled, not accidental: an empty answer scores 100 with no claims; empty
125
+ sources score 0 with every claim unsupported; an answer identical to a source scores 100;
126
+ the score is always between 0 and 100.
127
+
128
+ ## CLI
129
+
130
+ ```
131
+ hallucination-check ANSWER [-s SOURCE ...] [--threshold 0.55]
132
+ [--granularity sentence|clause|paragraph] [--no-fact-flags]
133
+ [--json] [--output PATH] [--fail-under SCORE]
134
+ ```
135
+
136
+ `ANSWER` and each `SOURCE` may be a file path or the text itself; `-` reads stdin. A value
137
+ that can only have meant a file -- one bare word ending in `.txt`, `.md`, `.json`, `.jsonl`
138
+ or `.ndjson` -- is an error when no such file exists, rather than being checked as literal
139
+ text, so a mistyped path never turns into a confident wrong report.
140
+
141
+ - `hallucination-check answer.txt -s passage1.txt -s passage2.txt` prints the summary.
142
+ - `hallucination-check "Paris is the capital of France." -s "France's capital is Paris."`
143
+ takes text directly.
144
+ - `cat answer.txt | hallucination-check - -s retrieved.jsonl --json` reads the answer from
145
+ stdin and the passages from a JSON-lines file (one object per line with `text` and an
146
+ optional `id`), and prints `to_dict()` as JSON.
147
+ - `--output PATH` writes that JSON to a file; `--fail-under 80` exits with status 2 when the
148
+ score is below 80, which is the useful thing to put in CI.
149
+
150
+ ## License
151
+
152
+ MIT
@@ -0,0 +1,50 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "hallucination-check"
7
+ version = "0.1.0"
8
+ description = "Check an answer against the sources it claims to use and flag every unsupported sentence"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Pranay Mahendrakar" }]
14
+ keywords = [
15
+ "hallucination",
16
+ "grounding",
17
+ "rag",
18
+ "citation",
19
+ "fact-checking",
20
+ "llm-evaluation",
21
+ "attribution",
22
+ "faithfulness",
23
+ "nlp",
24
+ ]
25
+ classifiers = [
26
+ "Development Status :: 4 - Beta",
27
+ "Intended Audience :: Developers",
28
+ "Intended Audience :: Science/Research",
29
+ "Programming Language :: Python :: 3",
30
+ "Programming Language :: Python :: 3 :: Only",
31
+ "Operating System :: OS Independent",
32
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
33
+ "Topic :: Text Processing :: Linguistic",
34
+ ]
35
+ dependencies = [
36
+ "numpy>=1.23",
37
+ ]
38
+
39
+ [project.optional-dependencies]
40
+ dev = ["pytest>=7"]
41
+
42
+ [project.scripts]
43
+ hallucination-check = "hallucination_check.cli:main"
44
+
45
+ [project.urls]
46
+ Homepage = "https://pypi.org/project/hallucination-check/"
47
+ Author = "https://pypi.org/user/pranaymahendrakar/"
48
+
49
+ [tool.hatch.build.targets.wheel]
50
+ packages = ["src/hallucination_check"]
@@ -0,0 +1,21 @@
1
+ """hallucination-check: check an answer against the sources it claims to use.
2
+
3
+ Quick use::
4
+
5
+ import hallucination_check as hc
6
+ report = hc.check(answer, sources)
7
+ print(report.summary())
8
+ print(report.unsupported)
9
+ """
10
+ from ._core import Claim, GroundingChecker, GroundingReport, check, check_batch
11
+
12
+ __version__ = "0.1.0"
13
+
14
+ __all__ = [
15
+ "Claim",
16
+ "GroundingChecker",
17
+ "GroundingReport",
18
+ "check",
19
+ "check_batch",
20
+ "__version__",
21
+ ]