hallucination-check 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hallucination_check-0.1.0/.gitignore +30 -0
- hallucination_check-0.1.0/LICENSE +21 -0
- hallucination_check-0.1.0/PKG-INFO +176 -0
- hallucination_check-0.1.0/README.md +152 -0
- hallucination_check-0.1.0/pyproject.toml +50 -0
- hallucination_check-0.1.0/src/hallucination_check/__init__.py +21 -0
- hallucination_check-0.1.0/src/hallucination_check/_core.py +647 -0
- hallucination_check-0.1.0/src/hallucination_check/_facts.py +534 -0
- hallucination_check-0.1.0/src/hallucination_check/_match.py +230 -0
- hallucination_check-0.1.0/src/hallucination_check/_text.py +341 -0
- hallucination_check-0.1.0/src/hallucination_check/cli.py +189 -0
- hallucination_check-0.1.0/tests/test_api.py +229 -0
- hallucination_check-0.1.0/tests/test_cli.py +164 -0
- hallucination_check-0.1.0/tests/test_edge_cases.py +176 -0
- hallucination_check-0.1.0/tests/test_facts.py +140 -0
- hallucination_check-0.1.0/tests/test_regressions.py +216 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Virtual environments and scratch space used while building. Never committed:
|
|
2
|
+
# they are large, machine-specific, and rebuilt from pyproject.toml anyway.
|
|
3
|
+
_venvs/
|
|
4
|
+
_proof/
|
|
5
|
+
.venv*/
|
|
6
|
+
|
|
7
|
+
# Build output. Wheels are built by the release workflow from this source, so a
|
|
8
|
+
# wheel in git could silently differ from the code beside it.
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
src/*.egg-info/
|
|
13
|
+
|
|
14
|
+
# Python noise
|
|
15
|
+
__pycache__/
|
|
16
|
+
*.py[cod]
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.mypy_cache/
|
|
19
|
+
.ruff_cache/
|
|
20
|
+
|
|
21
|
+
# Local verification state, not source
|
|
22
|
+
verify_results.json
|
|
23
|
+
publish-log.txt
|
|
24
|
+
published.json
|
|
25
|
+
|
|
26
|
+
# Credentials. None of these belong here, and this line is the backstop.
|
|
27
|
+
.pypirc
|
|
28
|
+
.env
|
|
29
|
+
*.pem
|
|
30
|
+
*.key
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pranay Mahendrakar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: hallucination-check
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Check an answer against the sources it claims to use and flag every unsupported sentence
|
|
5
|
+
Project-URL: Homepage, https://pypi.org/project/hallucination-check/
|
|
6
|
+
Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
|
|
7
|
+
Author: Pranay Mahendrakar
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: attribution,citation,fact-checking,faithfulness,grounding,hallucination,llm-evaluation,nlp,rag
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Requires-Dist: numpy>=1.23
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# hallucination-check
|
|
26
|
+
|
|
27
|
+
Check an answer against the sources it claims to use, and get back the exact sentences the
|
|
28
|
+
sources do not support.
|
|
29
|
+
|
|
30
|
+
## Install
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
pip install hallucination-check
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Only dependency: numpy. Nothing is downloaded at import or at run time.
|
|
37
|
+
|
|
38
|
+
## Quickstart
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
import hallucination_check as hc
|
|
42
|
+
|
|
43
|
+
answer = "The Eiffel Tower is in Paris. It was completed in 1889. It cost 42 million francs."
|
|
44
|
+
sources = ["The Eiffel Tower stands in Paris, France, and was completed in 1889."]
|
|
45
|
+
report = hc.check(answer, sources)
|
|
46
|
+
print(report.summary()) # 66.7 / 100 grounded, 1 claim flagged
|
|
47
|
+
print([c.text for c in report.unsupported]) # ['It cost 42 million francs.']
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
`sources` can be one string, a list of strings, or a list of dicts with `"text"` and an
|
|
51
|
+
optional `"id"`, so retrieved passages go straight in:
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
hc.check(answer, [{"id": "wiki:eiffel#3", "text": "..."}, {"id": "wiki:paris#1", "text": "..."}])
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## What it checks
|
|
58
|
+
|
|
59
|
+
- The answer is split into claims (one sentence each by default; `granularity="clause"` or
|
|
60
|
+
`"paragraph"` also work). Every claim is checked on its own.
|
|
61
|
+
- For each claim the sources are searched for the best-matching span of one to three
|
|
62
|
+
consecutive sentences. That span, its source id and the score are on the claim, so every
|
|
63
|
+
verdict can be traced back to the text it came from.
|
|
64
|
+
- The support score blends three lexical signals: character 4-gram cosine, word
|
|
65
|
+
unigram/bigram cosine, and how much of the claim's content vocabulary the span covers
|
|
66
|
+
(light plural and tense stripping, so "site" matches "sites"). A claim scoring at or above
|
|
67
|
+
`threshold` (default 0.55) counts as supported.
|
|
68
|
+
- **A claim carrying a number, date, name or quantity that appears nowhere in the sources is
|
|
69
|
+
flagged whatever its similarity**, because those are what people actually get wrong.
|
|
70
|
+
`12%`, `$4.3 billion`, `1,200`, `twenty-three`, `12 March 2024`, `November 2022` and
|
|
71
|
+
`NASA` are all recognized; numbers written in words and in digits compare equal.
|
|
72
|
+
- A fact has to appear in the *same role*, not merely somewhere in the pile of passages.
|
|
73
|
+
A measurement is tied to the unit or head noun after it and a name to the words around
|
|
74
|
+
it, so "reduced symptoms by 1,200 percent" is not backed up by "enrolled 1,200
|
|
75
|
+
participants" in another passage, "11 metres" is not backed up by the 11 in "Apollo 11",
|
|
76
|
+
and "developed at Pfizer" is not backed up by a passage that only mentions Pfizer's
|
|
77
|
+
earnings. A possessive in the source counts as the bare name, so "France's capital"
|
|
78
|
+
grounds a claim about France.
|
|
79
|
+
- When the best span states a *different* number or date for otherwise the same sentence,
|
|
80
|
+
the claim also lands in `report.contradictions` with a reason like
|
|
81
|
+
`the sources say 1,200 where the answer says 4,200`.
|
|
82
|
+
- A capitalized word at the start of a sentence is not treated as a name unless it is an
|
|
83
|
+
acronym or the start of a multi-word name, so "Cats sleep a lot." is not flagged for
|
|
84
|
+
"Cats".
|
|
85
|
+
- Everything is deterministic: the same answer and sources always give the same report.
|
|
86
|
+
Nothing is written, nothing is fetched, and the library never prints.
|
|
87
|
+
|
|
88
|
+
**Lexical grounding is a signal, not proof.** A claim can score well while quietly reversing
|
|
89
|
+
the source's meaning, and a correct paraphrase that shares no vocabulary will score badly.
|
|
90
|
+
The `embed` hook is how you strengthen it: pass any function that turns a list of strings
|
|
91
|
+
into vectors, and semantic similarity is blended into the score at equal weight (this
|
|
92
|
+
package stays free of model dependencies, so the model stays yours).
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
from sentence_transformers import SentenceTransformer # your choice, not a dependency
|
|
96
|
+
model = SentenceTransformer("all-MiniLM-L6-v2")
|
|
97
|
+
report = hc.check(answer, sources, embed=lambda texts: model.encode(texts))
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
With `embed` supplied, every source sentence is embedded (up to `embed_max_sentences`), so
|
|
101
|
+
the hook also finds spans that share no words at all with the claim.
|
|
102
|
+
|
|
103
|
+
## API
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
hc.check(answer, sources, *, threshold=0.55, granularity="sentence", embed=None) -> GroundingReport
|
|
107
|
+
hc.check_batch(pairs, **kw) -> list[GroundingReport]
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
`pairs` is an iterable of `(answer, sources)` tuples, or of dicts with `"answer"` and
|
|
111
|
+
`"sources"` keys. `**kw` goes to `GroundingChecker`.
|
|
112
|
+
|
|
113
|
+
`GroundingReport`
|
|
114
|
+
|
|
115
|
+
- `.score` - 0-100, the share of the answer's claims the sources support
|
|
116
|
+
- `.claims` - `list[Claim]` in answer order
|
|
117
|
+
- `.unsupported` - the claims the sources do not back up
|
|
118
|
+
- `.contradictions` - unsupported claims where a source states a different number or date
|
|
119
|
+
- `.citations` - `dict[claim index -> source id]` for the supported claims
|
|
120
|
+
- `.n_claims`, `.n_supported`, `.threshold`, `.granularity`, `.source_ids`, `.warnings`
|
|
121
|
+
- `.summary()` - human-readable text; `.to_dict()` - JSON-safe dict
|
|
122
|
+
|
|
123
|
+
`Claim(text, supported, confidence, best_source, best_span, reason)`
|
|
124
|
+
|
|
125
|
+
- `.text` - the claim as it appears in the answer
|
|
126
|
+
- `.supported` - the verdict
|
|
127
|
+
- `.confidence` - 0-1 support score for this claim
|
|
128
|
+
- `.best_source` - id of the source the best span came from (`"s1"`, `"s2"`, ... unless you
|
|
129
|
+
gave ids), or `None` when nothing matched
|
|
130
|
+
- `.best_span` - the source text that matched best
|
|
131
|
+
- `.reason` - one plain sentence saying why, e.g.
|
|
132
|
+
`these appear nowhere in the sources: 42 million`
|
|
133
|
+
- `.to_dict()`
|
|
134
|
+
|
|
135
|
+
```python
|
|
136
|
+
report = hc.check(answer, sources)
|
|
137
|
+
report.score # 66.7
|
|
138
|
+
report.claims[1].best_span # 'The Eiffel Tower stands in Paris, France, and was completed in 1889.'
|
|
139
|
+
report.citations # {0: 's1', 1: 's1'}
|
|
140
|
+
report.to_dict()["unsupported"] # [2]
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
`GroundingChecker(threshold=0.55, granularity="sentence", embed=None, embed_weight=0.5,
|
|
144
|
+
flag_missing_facts=True, char_n=4, window=3, top_k=25, embed_top=5,
|
|
145
|
+
embed_max_sentences=2000)` is the class underneath, with `.check(answer, sources)` and
|
|
146
|
+
`.check_batch(pairs)`. Set `flag_missing_facts=False` to score on similarity alone.
|
|
147
|
+
|
|
148
|
+
Edge cases are settled, not accidental: an empty answer scores 100 with no claims; empty
|
|
149
|
+
sources score 0 with every claim unsupported; an answer identical to a source scores 100;
|
|
150
|
+
the score is always between 0 and 100.
|
|
151
|
+
|
|
152
|
+
## CLI
|
|
153
|
+
|
|
154
|
+
```
|
|
155
|
+
hallucination-check ANSWER [-s SOURCE ...] [--threshold 0.55]
|
|
156
|
+
[--granularity sentence|clause|paragraph] [--no-fact-flags]
|
|
157
|
+
[--json] [--output PATH] [--fail-under SCORE]
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
`ANSWER` and each `SOURCE` may be a file path or the text itself; `-` reads stdin. A value
|
|
161
|
+
that can only have meant a file -- one bare word ending in `.txt`, `.md`, `.json`, `.jsonl`
|
|
162
|
+
or `.ndjson` -- is an error when no such file exists, rather than being checked as literal
|
|
163
|
+
text, so a mistyped path never turns into a confident wrong report.
|
|
164
|
+
|
|
165
|
+
- `hallucination-check answer.txt -s passage1.txt -s passage2.txt` prints the summary.
|
|
166
|
+
- `hallucination-check "Paris is the capital of France." -s "France's capital is Paris."`
|
|
167
|
+
takes text directly.
|
|
168
|
+
- `cat answer.txt | hallucination-check - -s retrieved.jsonl --json` reads the answer from
|
|
169
|
+
stdin and the passages from a JSON-lines file (one object per line with `text` and an
|
|
170
|
+
optional `id`), and prints `to_dict()` as JSON.
|
|
171
|
+
- `--output PATH` writes that JSON to a file; `--fail-under 80` exits with status 2 when the
|
|
172
|
+
score is below 80, which is the useful thing to put in CI.
|
|
173
|
+
|
|
174
|
+
## License
|
|
175
|
+
|
|
176
|
+
MIT
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
# hallucination-check
|
|
2
|
+
|
|
3
|
+
Check an answer against the sources it claims to use, and get back the exact sentences the
|
|
4
|
+
sources do not support.
|
|
5
|
+
|
|
6
|
+
## Install
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
pip install hallucination-check
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Only dependency: numpy. Nothing is downloaded at import or at run time.
|
|
13
|
+
|
|
14
|
+
## Quickstart
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
import hallucination_check as hc
|
|
18
|
+
|
|
19
|
+
answer = "The Eiffel Tower is in Paris. It was completed in 1889. It cost 42 million francs."
|
|
20
|
+
sources = ["The Eiffel Tower stands in Paris, France, and was completed in 1889."]
|
|
21
|
+
report = hc.check(answer, sources)
|
|
22
|
+
print(report.summary()) # 66.7 / 100 grounded, 1 claim flagged
|
|
23
|
+
print([c.text for c in report.unsupported]) # ['It cost 42 million francs.']
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
`sources` can be one string, a list of strings, or a list of dicts with `"text"` and an
|
|
27
|
+
optional `"id"`, so retrieved passages go straight in:
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
hc.check(answer, [{"id": "wiki:eiffel#3", "text": "..."}, {"id": "wiki:paris#1", "text": "..."}])
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## What it checks
|
|
34
|
+
|
|
35
|
+
- The answer is split into claims (one sentence each by default; `granularity="clause"` or
|
|
36
|
+
`"paragraph"` also work). Every claim is checked on its own.
|
|
37
|
+
- For each claim the sources are searched for the best-matching span of one to three
|
|
38
|
+
consecutive sentences. That span, its source id and the score are on the claim, so every
|
|
39
|
+
verdict can be traced back to the text it came from.
|
|
40
|
+
- The support score blends three lexical signals: character 4-gram cosine, word
|
|
41
|
+
unigram/bigram cosine, and how much of the claim's content vocabulary the span covers
|
|
42
|
+
(light plural and tense stripping, so "site" matches "sites"). A claim scoring at or above
|
|
43
|
+
`threshold` (default 0.55) counts as supported.
|
|
44
|
+
- **A claim carrying a number, date, name or quantity that appears nowhere in the sources is
|
|
45
|
+
flagged whatever its similarity**, because those are what people actually get wrong.
|
|
46
|
+
`12%`, `$4.3 billion`, `1,200`, `twenty-three`, `12 March 2024`, `November 2022` and
|
|
47
|
+
`NASA` are all recognized; numbers written in words and in digits compare equal.
|
|
48
|
+
- A fact has to appear in the *same role*, not merely somewhere in the pile of passages.
|
|
49
|
+
A measurement is tied to the unit or head noun after it and a name to the words around
|
|
50
|
+
it, so "reduced symptoms by 1,200 percent" is not backed up by "enrolled 1,200
|
|
51
|
+
participants" in another passage, "11 metres" is not backed up by the 11 in "Apollo 11",
|
|
52
|
+
and "developed at Pfizer" is not backed up by a passage that only mentions Pfizer's
|
|
53
|
+
earnings. A possessive in the source counts as the bare name, so "France's capital"
|
|
54
|
+
grounds a claim about France.
|
|
55
|
+
- When the best span states a *different* number or date for otherwise the same sentence,
|
|
56
|
+
the claim also lands in `report.contradictions` with a reason like
|
|
57
|
+
`the sources say 1,200 where the answer says 4,200`.
|
|
58
|
+
- A capitalized word at the start of a sentence is not treated as a name unless it is an
|
|
59
|
+
acronym or the start of a multi-word name, so "Cats sleep a lot." is not flagged for
|
|
60
|
+
"Cats".
|
|
61
|
+
- Everything is deterministic: the same answer and sources always give the same report.
|
|
62
|
+
Nothing is written, nothing is fetched, and the library never prints.
|
|
63
|
+
|
|
64
|
+
**Lexical grounding is a signal, not proof.** A claim can score well while quietly reversing
|
|
65
|
+
the source's meaning, and a correct paraphrase that shares no vocabulary will score badly.
|
|
66
|
+
The `embed` hook is how you strengthen it: pass any function that turns a list of strings
|
|
67
|
+
into vectors, and semantic similarity is blended into the score at equal weight (this
|
|
68
|
+
package stays free of model dependencies, so the model stays yours).
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from sentence_transformers import SentenceTransformer # your choice, not a dependency
|
|
72
|
+
model = SentenceTransformer("all-MiniLM-L6-v2")
|
|
73
|
+
report = hc.check(answer, sources, embed=lambda texts: model.encode(texts))
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
With `embed` supplied, every source sentence is embedded (up to `embed_max_sentences`), so
|
|
77
|
+
the hook also finds spans that share no words at all with the claim.
|
|
78
|
+
|
|
79
|
+
## API
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
hc.check(answer, sources, *, threshold=0.55, granularity="sentence", embed=None) -> GroundingReport
|
|
83
|
+
hc.check_batch(pairs, **kw) -> list[GroundingReport]
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
`pairs` is an iterable of `(answer, sources)` tuples, or of dicts with `"answer"` and
|
|
87
|
+
`"sources"` keys. `**kw` goes to `GroundingChecker`.
|
|
88
|
+
|
|
89
|
+
`GroundingReport`
|
|
90
|
+
|
|
91
|
+
- `.score` - 0-100, the share of the answer's claims the sources support
|
|
92
|
+
- `.claims` - `list[Claim]` in answer order
|
|
93
|
+
- `.unsupported` - the claims the sources do not back up
|
|
94
|
+
- `.contradictions` - unsupported claims where a source states a different number or date
|
|
95
|
+
- `.citations` - `dict[claim index -> source id]` for the supported claims
|
|
96
|
+
- `.n_claims`, `.n_supported`, `.threshold`, `.granularity`, `.source_ids`, `.warnings`
|
|
97
|
+
- `.summary()` - human-readable text; `.to_dict()` - JSON-safe dict
|
|
98
|
+
|
|
99
|
+
`Claim(text, supported, confidence, best_source, best_span, reason)`
|
|
100
|
+
|
|
101
|
+
- `.text` - the claim as it appears in the answer
|
|
102
|
+
- `.supported` - the verdict
|
|
103
|
+
- `.confidence` - 0-1 support score for this claim
|
|
104
|
+
- `.best_source` - id of the source the best span came from (`"s1"`, `"s2"`, ... unless you
|
|
105
|
+
gave ids), or `None` when nothing matched
|
|
106
|
+
- `.best_span` - the source text that matched best
|
|
107
|
+
- `.reason` - one plain sentence saying why, e.g.
|
|
108
|
+
`these appear nowhere in the sources: 42 million`
|
|
109
|
+
- `.to_dict()`
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
report = hc.check(answer, sources)
|
|
113
|
+
report.score # 66.7
|
|
114
|
+
report.claims[1].best_span # 'The Eiffel Tower stands in Paris, France, and was completed in 1889.'
|
|
115
|
+
report.citations # {0: 's1', 1: 's1'}
|
|
116
|
+
report.to_dict()["unsupported"] # [2]
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
`GroundingChecker(threshold=0.55, granularity="sentence", embed=None, embed_weight=0.5,
|
|
120
|
+
flag_missing_facts=True, char_n=4, window=3, top_k=25, embed_top=5,
|
|
121
|
+
embed_max_sentences=2000)` is the class underneath, with `.check(answer, sources)` and
|
|
122
|
+
`.check_batch(pairs)`. Set `flag_missing_facts=False` to score on similarity alone.
|
|
123
|
+
|
|
124
|
+
Edge cases are settled, not accidental: an empty answer scores 100 with no claims; empty
|
|
125
|
+
sources score 0 with every claim unsupported; an answer identical to a source scores 100;
|
|
126
|
+
the score is always between 0 and 100.
|
|
127
|
+
|
|
128
|
+
## CLI
|
|
129
|
+
|
|
130
|
+
```
|
|
131
|
+
hallucination-check ANSWER [-s SOURCE ...] [--threshold 0.55]
|
|
132
|
+
[--granularity sentence|clause|paragraph] [--no-fact-flags]
|
|
133
|
+
[--json] [--output PATH] [--fail-under SCORE]
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
`ANSWER` and each `SOURCE` may be a file path or the text itself; `-` reads stdin. A value
|
|
137
|
+
that can only have meant a file -- one bare word ending in `.txt`, `.md`, `.json`, `.jsonl`
|
|
138
|
+
or `.ndjson` -- is an error when no such file exists, rather than being checked as literal
|
|
139
|
+
text, so a mistyped path never turns into a confident wrong report.
|
|
140
|
+
|
|
141
|
+
- `hallucination-check answer.txt -s passage1.txt -s passage2.txt` prints the summary.
|
|
142
|
+
- `hallucination-check "Paris is the capital of France." -s "France's capital is Paris."`
|
|
143
|
+
takes text directly.
|
|
144
|
+
- `cat answer.txt | hallucination-check - -s retrieved.jsonl --json` reads the answer from
|
|
145
|
+
stdin and the passages from a JSON-lines file (one object per line with `text` and an
|
|
146
|
+
optional `id`), and prints `to_dict()` as JSON.
|
|
147
|
+
- `--output PATH` writes that JSON to a file; `--fail-under 80` exits with status 2 when the
|
|
148
|
+
score is below 80, which is the useful thing to put in CI.
|
|
149
|
+
|
|
150
|
+
## License
|
|
151
|
+
|
|
152
|
+
MIT
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "hallucination-check"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Check an answer against the sources it claims to use and flag every unsupported sentence"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Pranay Mahendrakar" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"hallucination",
|
|
16
|
+
"grounding",
|
|
17
|
+
"rag",
|
|
18
|
+
"citation",
|
|
19
|
+
"fact-checking",
|
|
20
|
+
"llm-evaluation",
|
|
21
|
+
"attribution",
|
|
22
|
+
"faithfulness",
|
|
23
|
+
"nlp",
|
|
24
|
+
]
|
|
25
|
+
classifiers = [
|
|
26
|
+
"Development Status :: 4 - Beta",
|
|
27
|
+
"Intended Audience :: Developers",
|
|
28
|
+
"Intended Audience :: Science/Research",
|
|
29
|
+
"Programming Language :: Python :: 3",
|
|
30
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
31
|
+
"Operating System :: OS Independent",
|
|
32
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
33
|
+
"Topic :: Text Processing :: Linguistic",
|
|
34
|
+
]
|
|
35
|
+
dependencies = [
|
|
36
|
+
"numpy>=1.23",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
[project.optional-dependencies]
|
|
40
|
+
dev = ["pytest>=7"]
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
hallucination-check = "hallucination_check.cli:main"
|
|
44
|
+
|
|
45
|
+
[project.urls]
|
|
46
|
+
Homepage = "https://pypi.org/project/hallucination-check/"
|
|
47
|
+
Author = "https://pypi.org/user/pranaymahendrakar/"
|
|
48
|
+
|
|
49
|
+
[tool.hatch.build.targets.wheel]
|
|
50
|
+
packages = ["src/hallucination_check"]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""hallucination-check: check an answer against the sources it claims to use.
|
|
2
|
+
|
|
3
|
+
Quick use::
|
|
4
|
+
|
|
5
|
+
import hallucination_check as hc
|
|
6
|
+
report = hc.check(answer, sources)
|
|
7
|
+
print(report.summary())
|
|
8
|
+
print(report.unsupported)
|
|
9
|
+
"""
|
|
10
|
+
from ._core import Claim, GroundingChecker, GroundingReport, check, check_batch
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.0"
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"Claim",
|
|
16
|
+
"GroundingChecker",
|
|
17
|
+
"GroundingReport",
|
|
18
|
+
"check",
|
|
19
|
+
"check_batch",
|
|
20
|
+
"__version__",
|
|
21
|
+
]
|