claimlens 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- claimlens-1.0.0/.env-example +4 -0
- claimlens-1.0.0/.gitattributes +8 -0
- claimlens-1.0.0/.gitignore +44 -0
- claimlens-1.0.0/.python-version +1 -0
- claimlens-1.0.0/.vscode/launch.json +224 -0
- claimlens-1.0.0/LICENSE +21 -0
- claimlens-1.0.0/PKG-INFO +342 -0
- claimlens-1.0.0/README.md +309 -0
- claimlens-1.0.0/architecture.md +269 -0
- claimlens-1.0.0/pyproject.toml +60 -0
- claimlens-1.0.0/src/claimlens/__init__.py +167 -0
- claimlens-1.0.0/src/claimlens/cli.py +715 -0
- claimlens-1.0.0/src/claimlens/eval/__init__.py +7 -0
- claimlens-1.0.0/src/claimlens/eval/base.py +111 -0
- claimlens-1.0.0/src/claimlens/eval/checkereval.py +693 -0
- claimlens-1.0.0/src/claimlens/eval/extractoreval.py +1326 -0
- claimlens-1.0.0/src/claimlens/eval/metrics.py +293 -0
- claimlens-1.0.0/src/claimlens/exceptions.py +93 -0
- claimlens-1.0.0/src/claimlens/llmclient.py +1068 -0
- claimlens-1.0.0/src/claimlens/models.py +192 -0
- claimlens-1.0.0/src/claimlens/pipelines/__init__.py +18 -0
- claimlens-1.0.0/src/claimlens/pipelines/directions.py +301 -0
- claimlens-1.0.0/src/claimlens/pipelines/faithfulness.py +620 -0
- claimlens-1.0.0/src/claimlens/pipelines/ragchecker.py +1096 -0
- claimlens-1.0.0/src/claimlens/pipelines/refchecker.py +363 -0
- claimlens-1.0.0/src/claimlens/prompt_map.json +12 -0
- claimlens-1.0.0/src/claimlens/services/atomization.py +441 -0
- claimlens-1.0.0/src/claimlens/services/base.py +244 -0
- claimlens-1.0.0/src/claimlens/services/checking.py +671 -0
- claimlens-1.0.0/src/claimlens/services/extraction.py +398 -0
- claimlens-1.0.0/src/claimlens/settings.py +161 -0
- claimlens-1.0.0/src/claimlens/stats.py +696 -0
- claimlens-1.0.0/src/claimlens/templates/common.css +52 -0
- claimlens-1.0.0/src/claimlens/templates/common.js +67 -0
- claimlens-1.0.0/src/claimlens/templates/ragcheck.html +289 -0
- claimlens-1.0.0/src/claimlens/utils.py +288 -0
- claimlens-1.0.0/src/claimlens/viewer.py +56 -0
- claimlens-1.0.0/src/claimlens/workers/atomizer.py +318 -0
- claimlens-1.0.0/src/claimlens/workers/checker.py +574 -0
- claimlens-1.0.0/src/claimlens/workers/extractor.py +288 -0
- claimlens-1.0.0/tests/conftest.py +18 -0
- claimlens-1.0.0/tests/integration/test_finish_reason_errors.py +256 -0
- claimlens-1.0.0/tests/integration/test_llmclient_fatal.py +484 -0
- claimlens-1.0.0/tests/integration/test_llmclient_retry.py +363 -0
- claimlens-1.0.0/tests/integration/test_llmclient_timeout.py +157 -0
- claimlens-1.0.0/tests/integration/test_timeout_end_to_end.py +106 -0
- claimlens-1.0.0/tests/unit/test_abstention.py +68 -0
- claimlens-1.0.0/tests/unit/test_atomization_service.py +355 -0
- claimlens-1.0.0/tests/unit/test_atomization_validate_filter.py +208 -0
- claimlens-1.0.0/tests/unit/test_atomizer_worker.py +287 -0
- claimlens-1.0.0/tests/unit/test_checkereval.py +595 -0
- claimlens-1.0.0/tests/unit/test_checking_service.py +731 -0
- claimlens-1.0.0/tests/unit/test_cli_output.py +117 -0
- claimlens-1.0.0/tests/unit/test_directions.py +244 -0
- claimlens-1.0.0/tests/unit/test_extraction_service.py +391 -0
- claimlens-1.0.0/tests/unit/test_extractoreval.py +968 -0
- claimlens-1.0.0/tests/unit/test_facades.py +138 -0
- claimlens-1.0.0/tests/unit/test_faithfulness.py +201 -0
- claimlens-1.0.0/tests/unit/test_llmclient_strategy_cache.py +119 -0
- claimlens-1.0.0/tests/unit/test_meta_request_strategies.py +163 -0
- claimlens-1.0.0/tests/unit/test_meta_usage.py +120 -0
- claimlens-1.0.0/tests/unit/test_metrics.py +182 -0
- claimlens-1.0.0/tests/unit/test_normalization.py +132 -0
- claimlens-1.0.0/tests/unit/test_output_conventions.py +399 -0
- claimlens-1.0.0/tests/unit/test_plain_prompts.py +171 -0
- claimlens-1.0.0/tests/unit/test_ragchecker.py +743 -0
- claimlens-1.0.0/tests/unit/test_refchecker_pipeline.py +276 -0
- claimlens-1.0.0/tests/unit/test_report_envelope.py +161 -0
- claimlens-1.0.0/tests/unit/test_runs.py +195 -0
- claimlens-1.0.0/tests/unit/test_schema_shape.py +124 -0
- claimlens-1.0.0/tests/unit/test_timeout_downstream.py +112 -0
- claimlens-1.0.0/tests/unit/test_utils.py +197 -0
- claimlens-1.0.0/tests/unit/test_viewer.py +163 -0
- claimlens-1.0.0/uv.lock +1734 -0
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# Normalize line endings: source stays LF in the repo AND in the working
|
|
2
|
+
# tree, so Windows editors stop flip-flopping and git stops warning
|
|
3
|
+
# "LF will be replaced by CRLF" on every command.
|
|
4
|
+
* text=auto
|
|
5
|
+
*.py text eol=lf
|
|
6
|
+
*.md text eol=lf
|
|
7
|
+
*.toml text eol=lf
|
|
8
|
+
*.json text eol=lf
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# ── Build / Dist ─────────────────────────────────────────────────────────────
|
|
2
|
+
dist/
|
|
3
|
+
build/
|
|
4
|
+
*.egg-info/
|
|
5
|
+
|
|
6
|
+
# ── Python ───────────────────────────────────────────────────────────────────
|
|
7
|
+
__pycache__/
|
|
8
|
+
*.py[cod]
|
|
9
|
+
*.pyo
|
|
10
|
+
|
|
11
|
+
# ── Virtual Environments ────────────────────────────────────────────────────
|
|
12
|
+
.venv/
|
|
13
|
+
venv/
|
|
14
|
+
|
|
15
|
+
# ── IDE ──────────────────────────────────────────────────────────────────────
|
|
16
|
+
# launch.json is shared: it documents the standard workflows. Everything else
|
|
17
|
+
# under .vscode/ is personal.
|
|
18
|
+
.vscode/*
|
|
19
|
+
!.vscode/launch.json
|
|
20
|
+
.idea/
|
|
21
|
+
|
|
22
|
+
# ── Test / Coverage ─────────────────────────────────────────────────────────
|
|
23
|
+
.pytest_cache/
|
|
24
|
+
.coverage
|
|
25
|
+
htmlcov/
|
|
26
|
+
|
|
27
|
+
# ── Secrets / Env ────────────────────────────────────────────────────────────
|
|
28
|
+
*.env
|
|
29
|
+
.env*
|
|
30
|
+
!.env-example
|
|
31
|
+
|
|
32
|
+
# ── OS ───────────────────────────────────────────────────────────────────────
|
|
33
|
+
.DS_Store
|
|
34
|
+
Thumbs.db
|
|
35
|
+
|
|
36
|
+
# ── Results (local runs) ────────────────────────────────────────────────────
|
|
37
|
+
# Outputs land next to their input, so examples/ and eval_data/ grow their own.
|
|
38
|
+
**/results/
|
|
39
|
+
|
|
40
|
+
ignore-found-bugs.md
|
|
41
|
+
|
|
42
|
+
agent.md
|
|
43
|
+
|
|
44
|
+
eval_cases_reference.md
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.12
|
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": "0.2.0",
|
|
3
|
+
"inputs": [
|
|
4
|
+
{
|
|
5
|
+
"id": "extractorModel",
|
|
6
|
+
"type": "promptString",
|
|
7
|
+
"description": "Extractor model — bare id when a base URL is set (e.g. llama3.2:3b, openai/gpt-5.6-luna)",
|
|
8
|
+
"default": "z-ai/glm-5.3-flash"
|
|
9
|
+
},
|
|
10
|
+
{
|
|
11
|
+
"id": "extractorBaseApi",
|
|
12
|
+
"type": "promptString",
|
|
13
|
+
"description": "Extractor base URL — leave blank for LiteLLM provider routing (then prefix the model, e.g. openrouter/...)",
|
|
14
|
+
"default": "https://openrouter.ai/api/v1"
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
"id": "checkerModel",
|
|
18
|
+
"type": "promptString",
|
|
19
|
+
"description": "Checker model — bare id when a base URL is set (e.g. llama3.2:3b, openai/gpt-5.6-luna)",
|
|
20
|
+
"default": "z-ai/glm-5.3-flash"
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
"id": "checkerBaseApi",
|
|
24
|
+
"type": "promptString",
|
|
25
|
+
"description": "Checker base URL — leave blank for LiteLLM provider routing (then prefix the model, e.g. openrouter/...)",
|
|
26
|
+
"default": "https://openrouter.ai/api/v1"
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"id": "atomizerModel",
|
|
30
|
+
"type": "promptString",
|
|
31
|
+
"description": "Atomizer model — measures whether extracted claims are atomic. Needs ATOMIZER_API_KEY.",
|
|
32
|
+
"default": "z-ai/glm-5.3-flash"
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"id": "atomizerBaseApi",
|
|
36
|
+
"type": "promptString",
|
|
37
|
+
"description": "Atomizer base URL — leave blank for LiteLLM provider routing",
|
|
38
|
+
"default": "https://openrouter.ai/api/v1"
|
|
39
|
+
}
|
|
40
|
+
],
|
|
41
|
+
"configurations": [
|
|
42
|
+
{
|
|
43
|
+
"name": "01 · extract — kepler22b",
|
|
44
|
+
"type": "debugpy",
|
|
45
|
+
"request": "launch",
|
|
46
|
+
"program": "${workspaceFolder}/.venv/bin/claimlens",
|
|
47
|
+
"args": [
|
|
48
|
+
"extract",
|
|
49
|
+
"examples/extract/kepler22b.json",
|
|
50
|
+
"--extractor-model",
|
|
51
|
+
"${input:extractorModel}",
|
|
52
|
+
"--extractor-base-api",
|
|
53
|
+
"${input:extractorBaseApi}"
|
|
54
|
+
],
|
|
55
|
+
"python": "${workspaceFolder}/.venv/bin/python",
|
|
56
|
+
"cwd": "${workspaceFolder}",
|
|
57
|
+
"console": "integratedTerminal",
|
|
58
|
+
"justMyCode": false,
|
|
59
|
+
"presentation": {
|
|
60
|
+
"group": "1 · Walkthrough",
|
|
61
|
+
"order": 1
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"name": "02 · check — kepler22b (extracted)",
|
|
66
|
+
"type": "debugpy",
|
|
67
|
+
"request": "launch",
|
|
68
|
+
"program": "${workspaceFolder}/.venv/bin/claimlens",
|
|
69
|
+
"args": [
|
|
70
|
+
"check",
|
|
71
|
+
"examples/check/kepler22b_extracted.json",
|
|
72
|
+
"--extractor-model",
|
|
73
|
+
"openai/gpt-5.6-luna",
|
|
74
|
+
"--checker-model",
|
|
75
|
+
"${input:checkerModel}",
|
|
76
|
+
"--checker-base-api",
|
|
77
|
+
"${input:checkerBaseApi}"
|
|
78
|
+
],
|
|
79
|
+
"python": "${workspaceFolder}/.venv/bin/python",
|
|
80
|
+
"cwd": "${workspaceFolder}",
|
|
81
|
+
"console": "integratedTerminal",
|
|
82
|
+
"justMyCode": false,
|
|
83
|
+
"presentation": {
|
|
84
|
+
"group": "1 · Walkthrough",
|
|
85
|
+
"order": 2
|
|
86
|
+
}
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"name": "03 · refcheck — kepler22b",
|
|
90
|
+
"type": "debugpy",
|
|
91
|
+
"request": "launch",
|
|
92
|
+
"program": "${workspaceFolder}/.venv/bin/claimlens",
|
|
93
|
+
"args": [
|
|
94
|
+
"refcheck",
|
|
95
|
+
"examples/refcheck/kepler22b.json",
|
|
96
|
+
"--extractor-model",
|
|
97
|
+
"${input:extractorModel}",
|
|
98
|
+
"--extractor-base-api",
|
|
99
|
+
"${input:extractorBaseApi}",
|
|
100
|
+
"--checker-model",
|
|
101
|
+
"${input:checkerModel}",
|
|
102
|
+
"--checker-base-api",
|
|
103
|
+
"${input:checkerBaseApi}",
|
|
104
|
+
"--debug"
|
|
105
|
+
],
|
|
106
|
+
"python": "${workspaceFolder}/.venv/bin/python",
|
|
107
|
+
"cwd": "${workspaceFolder}",
|
|
108
|
+
"console": "integratedTerminal",
|
|
109
|
+
"justMyCode": false,
|
|
110
|
+
"presentation": {
|
|
111
|
+
"group": "1 · Walkthrough",
|
|
112
|
+
"order": 3
|
|
113
|
+
}
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
"name": "04 · faithcheck — kepler22b",
|
|
117
|
+
"type": "debugpy",
|
|
118
|
+
"request": "launch",
|
|
119
|
+
"program": "${workspaceFolder}/.venv/bin/claimlens",
|
|
120
|
+
"args": [
|
|
121
|
+
"faithcheck",
|
|
122
|
+
"examples/faithcheck/kepler22b.json",
|
|
123
|
+
"--extractor-model",
|
|
124
|
+
"${input:extractorModel}",
|
|
125
|
+
"--extractor-base-api",
|
|
126
|
+
"${input:extractorBaseApi}",
|
|
127
|
+
"--checker-model",
|
|
128
|
+
"${input:checkerModel}",
|
|
129
|
+
"--checker-base-api",
|
|
130
|
+
"${input:checkerBaseApi}",
|
|
131
|
+
"--runs",
|
|
132
|
+
"2"
|
|
133
|
+
],
|
|
134
|
+
"python": "${workspaceFolder}/.venv/bin/python",
|
|
135
|
+
"cwd": "${workspaceFolder}",
|
|
136
|
+
"console": "integratedTerminal",
|
|
137
|
+
"justMyCode": false,
|
|
138
|
+
"presentation": {
|
|
139
|
+
"group": "1 · Walkthrough",
|
|
140
|
+
"order": 4
|
|
141
|
+
}
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
"name": "05 · ragcheck — kepler22b",
|
|
145
|
+
"type": "debugpy",
|
|
146
|
+
"request": "launch",
|
|
147
|
+
"program": "${workspaceFolder}/.venv/bin/claimlens",
|
|
148
|
+
"args": [
|
|
149
|
+
"ragcheck",
|
|
150
|
+
"examples/ragcheck/kepler22b.json",
|
|
151
|
+
"--extractor-model",
|
|
152
|
+
"${input:extractorModel}",
|
|
153
|
+
"--extractor-base-api",
|
|
154
|
+
"${input:extractorBaseApi}",
|
|
155
|
+
"--checker-model",
|
|
156
|
+
"${input:checkerModel}",
|
|
157
|
+
"--checker-base-api",
|
|
158
|
+
"${input:checkerBaseApi}",
|
|
159
|
+
"--runs",
|
|
160
|
+
"2"
|
|
161
|
+
],
|
|
162
|
+
"python": "${workspaceFolder}/.venv/bin/python",
|
|
163
|
+
"cwd": "${workspaceFolder}",
|
|
164
|
+
"console": "integratedTerminal",
|
|
165
|
+
"justMyCode": false,
|
|
166
|
+
"presentation": {
|
|
167
|
+
"group": "1 · Walkthrough",
|
|
168
|
+
"order": 5
|
|
169
|
+
}
|
|
170
|
+
},
|
|
171
|
+
{
|
|
172
|
+
"type": "debugpy",
|
|
173
|
+
"request": "launch",
|
|
174
|
+
"program": "${workspaceFolder}/.venv/bin/claimlens",
|
|
175
|
+
"python": "${workspaceFolder}/.venv/bin/python",
|
|
176
|
+
"cwd": "${workspaceFolder}",
|
|
177
|
+
"console": "integratedTerminal",
|
|
178
|
+
"justMyCode": false,
|
|
179
|
+
"name": "01 · eval checker — msmarco_gpt4_5‚",
|
|
180
|
+
"args": [
|
|
181
|
+
"eval",
|
|
182
|
+
"checker",
|
|
183
|
+
"eval_data/msmarco/msmarco_gpt4_25.json",
|
|
184
|
+
"--checker-model",
|
|
185
|
+
"${input:checkerModel}",
|
|
186
|
+
"--checker-base-api",
|
|
187
|
+
"${input:checkerBaseApi}",
|
|
188
|
+
"--runs",
|
|
189
|
+
"1"
|
|
190
|
+
],
|
|
191
|
+
"presentation": {
|
|
192
|
+
"group": "2 · Evaluation",
|
|
193
|
+
"order": 1
|
|
194
|
+
}
|
|
195
|
+
},
|
|
196
|
+
{
|
|
197
|
+
"type": "debugpy",
|
|
198
|
+
"request": "launch",
|
|
199
|
+
"program": "${workspaceFolder}/.venv/bin/claimlens",
|
|
200
|
+
"python": "${workspaceFolder}/.venv/bin/python",
|
|
201
|
+
"cwd": "${workspaceFolder}",
|
|
202
|
+
"console": "integratedTerminal",
|
|
203
|
+
"justMyCode": false,
|
|
204
|
+
"name": "02 · eval extractor — msmarco_gpt4_25",
|
|
205
|
+
"args": [
|
|
206
|
+
"eval",
|
|
207
|
+
"extractor",
|
|
208
|
+
"eval_data/msmarco/msmarco_gpt4_5.json",
|
|
209
|
+
"--extractor-model",
|
|
210
|
+
"${input:extractorModel}",
|
|
211
|
+
"--extractor-base-api",
|
|
212
|
+
"${input:extractorBaseApi}",
|
|
213
|
+
"--checker-model",
|
|
214
|
+
"${input:checkerModel}",
|
|
215
|
+
"--checker-base-api",
|
|
216
|
+
"${input:checkerBaseApi}"
|
|
217
|
+
],
|
|
218
|
+
"presentation": {
|
|
219
|
+
"group": "2 · Evaluation",
|
|
220
|
+
"order": 2
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
]
|
|
224
|
+
}
|
claimlens-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Arthur Galas
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
claimlens-1.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,342 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: claimlens
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Claim-level evaluation for LLM outputs: decompose text into atomic claims, then verify every claim against a reference.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Arthur-g-p/claimlens
|
|
6
|
+
Project-URL: Repository, https://github.com/Arthur-g-p/claimlens
|
|
7
|
+
Project-URL: Issues, https://github.com/Arthur-g-p/claimlens/issues
|
|
8
|
+
Author: Arthur Galas
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: claim-verification,evaluation,faithfulness,hallucination,llm,rag
|
|
12
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Requires-Python: >=3.12
|
|
19
|
+
Requires-Dist: httpx>=0.27.0
|
|
20
|
+
Requires-Dist: openai>=1.0
|
|
21
|
+
Requires-Dist: pydantic>=2.0
|
|
22
|
+
Requires-Dist: python-dotenv>=1.0.0
|
|
23
|
+
Requires-Dist: tqdm>=4.60
|
|
24
|
+
Requires-Dist: typer>=0.9.0
|
|
25
|
+
Provides-Extra: litellm
|
|
26
|
+
Requires-Dist: litellm>=1.0.0; extra == 'litellm'
|
|
27
|
+
Provides-Extra: test
|
|
28
|
+
Requires-Dist: coverage>=7.0; extra == 'test'
|
|
29
|
+
Requires-Dist: litellm>=1.0.0; extra == 'test'
|
|
30
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'test'
|
|
31
|
+
Requires-Dist: pytest>=8.0; extra == 'test'
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# ClaimLens
|
|
35
|
+
|
|
36
|
+
Claim-level evaluation for LLM outputs: decompose text into atomic claims,
|
|
37
|
+
then verify every claim against a reference.
|
|
38
|
+
|
|
39
|
+
Instead of asking a judge LLM "rate this answer 1-10", ClaimLens
|
|
40
|
+
extracts atomic `(subject, predicate, object)` claims from a response with
|
|
41
|
+
an extractor LLM, then classifies each claim against a reference with a
|
|
42
|
+
checker LLM: Entailment, Contradiction, or Neutral. Every score the toolkit
|
|
43
|
+
produces is an aggregate over per-claim verdicts you can read, audit, and
|
|
44
|
+
disagree with.
|
|
45
|
+
|
|
46
|
+
## Commands
|
|
47
|
+
|
|
48
|
+
| Command | What it does |
|
|
49
|
+
| --- | --- |
|
|
50
|
+
| `ragcheck` | RAGChecker-style RAG evaluation: 2 extractions + 4 checking directions, one self-contained JSON report (precision, recall, F1, faithfulness, and claim-level detail) |
|
|
51
|
+
| `faithcheck` | Faithfulness checking **without ground truth**: response claims vs the retrieved context. Works on live traffic. |
|
|
52
|
+
| `refcheck` | Classic reference checking: extraction + checking in one pass |
|
|
53
|
+
| `extract` / `check` / `atomize` | The individual building blocks, runnable standalone — walked through in order under [examples/](examples/README.md) |
|
|
54
|
+
| `eval extractor` / `eval checker` | Meta-evaluation: measure the extractor and checker themselves against labeled data. Run `eval checker` first — `eval extractor` uses a checker, so an unqualified one makes its numbers partial |
|
|
55
|
+
|
|
56
|
+
## Why this instead of a single judge score
|
|
57
|
+
|
|
58
|
+
- **Auditable**: every metric decomposes into claim-level verdicts with
|
|
59
|
+
explanations. You can see exactly which fact failed and why.
|
|
60
|
+
- **Variance is first-class**: LLM-based metrics are stochastic. Pass
|
|
61
|
+
`--runs N` to `ragcheck`, `faithcheck`, `eval extractor` or
|
|
62
|
+
`eval checker` and get `mean +/- std [min, max]` per metric across N
|
|
63
|
+
full repetitions, plus each run's complete report. A single-run score
|
|
64
|
+
overstates your certainty.
|
|
65
|
+
- **The evaluator is itself evaluated**: `eval extractor` and
|
|
66
|
+
`eval checker` measure the measurement tool against ground-truth
|
|
67
|
+
labels, so you know how much to trust the numbers before you compare
|
|
68
|
+
systems with them.
|
|
69
|
+
- **Honest failure handling**: per-item errors (context too long, content
|
|
70
|
+
policy, parse failure) are recorded as values next to the affected claim,
|
|
71
|
+
never silently zeroed. A claim the checker failed to judge leaves both
|
|
72
|
+
sides of the ratio rather than counting against the system under test,
|
|
73
|
+
and a metric with an empty denominator is `null`, never `0.0`.
|
|
74
|
+
Abstentions ("I don't know") are first-class labeled outcomes: judged
|
|
75
|
+
justified or unjustified, their cause apportioned between retriever
|
|
76
|
+
and generator, and unanswerable questions (`"gt_answer": ""`) expose
|
|
77
|
+
unwarranted answers — while the metrics stay standard (a refusal is
|
|
78
|
+
non-delivery; see docs/ragchecker.md#abstention).
|
|
79
|
+
- **Any OpenAI-compatible endpoint**: point the extractor and checker at
|
|
80
|
+
different models, providers, or local servers independently.
|
|
81
|
+
|
|
82
|
+
## Installation
|
|
83
|
+
|
|
84
|
+
Requires Python 3.12+. The project uses [uv](https://docs.astral.sh/uv/) for
|
|
85
|
+
dependency management — install it first if you don't have it:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
# macOS / Linux
|
|
89
|
+
curl -LsSf https://astral.sh/uv/install.sh | sh
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
# macOS (Homebrew alternative)
|
|
94
|
+
brew install uv
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
```powershell
|
|
98
|
+
# Windows (PowerShell)
|
|
99
|
+
powershell -ExecutionPolicy ByPass -c "irm https://astral.sh/uv/install.ps1 | iex"
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Then clone and install into a virtual environment:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
git clone https://github.com/Arthur-g-p/claimlens.git
|
|
106
|
+
cd claimlens
|
|
107
|
+
uv venv
|
|
108
|
+
uv pip install -e .
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
`uv venv` reads `.python-version` and downloads CPython 3.12 if it isn't
|
|
112
|
+
already present. It does not touch your system Python.
|
|
113
|
+
|
|
114
|
+
Activate the environment before using the `claimlens` command:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
source .venv/bin/activate # macOS / Linux
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
```powershell
|
|
121
|
+
.venv\Scripts\activate # Windows
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Activation is per-shell — you need it again in every new terminal. To skip it,
|
|
125
|
+
prefix commands with `uv run` instead (`uv run claimlens --help`).
|
|
126
|
+
|
|
127
|
+
To also install the test dependencies (pytest, pytest-asyncio, coverage):
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
uv pip install -e ".[test]"
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
LiteLLM routing (a provider prefix on the model name, no base URL) is an
|
|
134
|
+
optional extra. The default install talks to any OpenAI-compatible endpoint
|
|
135
|
+
directly and stays small. To add the routing path:
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
uv pip install -e ".[litellm]"
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
From PyPI that is `pip install "claimlens[litellm]"`.
|
|
142
|
+
|
|
143
|
+
Verify the install:
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
claimlens --help
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## Quick start
|
|
150
|
+
|
|
151
|
+
Copy the environment template and fill in your API keys:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
cp .env-example .env
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
```
|
|
158
|
+
EXTRACTOR_API_KEY=...
|
|
159
|
+
CHECKER_API_KEY=...
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
Input is a JSON list of items. For `ragcheck`, each item needs `response`,
|
|
163
|
+
`gt_answer`, and `retrieved_context`:
|
|
164
|
+
|
|
165
|
+
```json
|
|
166
|
+
[
|
|
167
|
+
{
|
|
168
|
+
"question": "Who wrote Faust?",
|
|
169
|
+
"response": "Faust was written by Goethe in the 19th century.",
|
|
170
|
+
"gt_answer": "Johann Wolfgang von Goethe wrote Faust.",
|
|
171
|
+
"retrieved_context": ["Goethe published Faust, Part One in 1808. ..."]
|
|
172
|
+
}
|
|
173
|
+
]
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
The input file is a **positional** argument on every command — pass the path
|
|
177
|
+
bare, with no flag in front of it.
|
|
178
|
+
|
|
179
|
+
You do not need to prepare data to start. `examples/` ships a five-step
|
|
180
|
+
walkthrough over one dataset, where each step needs a little more than the last,
|
|
181
|
+
plus `eval_data/` with human-labelled data for the `eval` commands.
|
|
182
|
+
|
|
183
|
+
Start at step 01 — decompose responses into claims:
|
|
184
|
+
|
|
185
|
+
```bash
|
|
186
|
+
claimlens extract examples/extract/kepler22b.json --extractor-model gpt-4o-mini
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
Run the full RAG pipeline (step 05), which needs `gt_answer` and
|
|
190
|
+
`retrieved_context` as well:
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
claimlens ragcheck examples/ragcheck/kepler22b.json --extractor-model gpt-4o-mini --checker-model gpt-4o-mini
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
Repeat the whole experiment five times and report variance (5x LLM cost):
|
|
197
|
+
|
|
198
|
+
```bash
|
|
199
|
+
claimlens ragcheck examples/ragcheck/kepler22b.json --extractor-model gpt-4o-mini --checker-model gpt-4o-mini --runs 5
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
Check faithfulness with no ground truth at all (step 04):
|
|
203
|
+
|
|
204
|
+
```bash
|
|
205
|
+
claimlens faithcheck examples/faithcheck/kepler22b.json --extractor-model gpt-4o-mini --checker-model gpt-4o-mini
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
The model flags have short aliases (`-e`, `-c`, `-m`), but the long names are
|
|
209
|
+
spelled the same way in every command, so those are the ones worth learning.
|
|
210
|
+
[examples/README.md](examples/README.md) lists which fields and which arguments
|
|
211
|
+
each step needs.
|
|
212
|
+
|
|
213
|
+
The pipelines and evals (`ragcheck`, `faithcheck`, `refcheck`, `eval
|
|
214
|
+
extractor`, `eval checker`) write two JSON files. The record holds everything: an `_args` block (what
|
|
215
|
+
you asked for — every flag of the invocation), a `_meta` block (what the run
|
|
216
|
+
turned out to be — counts, timings, derived keys), `metrics` where the command
|
|
217
|
+
computes any, every count the console prints, and the complete per-item
|
|
218
|
+
claim record — `--runs N` adds entries, it never reshapes. The
|
|
219
|
+
`_findings.json` sibling is the review queue: only what went wrong, keyed by
|
|
220
|
+
the console's own branch names, derived from the record. The building
|
|
221
|
+
blocks (`extract`, `check`, `atomize`) instead emit the item list itself,
|
|
222
|
+
enriched in place, so each one's output is the next one's input.
|
|
223
|
+
|
|
224
|
+
`ragcheck` writes a third file next to those two: `{report_stem}.html`, a
|
|
225
|
+
self-contained viewer of the same record. One screen for the run — the main
|
|
226
|
+
metrics with their spread over `--runs`, then every item as a row of dots,
|
|
227
|
+
one per response claim, lime when a chunk grounds it — and one screen per
|
|
228
|
+
item: GT claims, retrieved chunks and response claims in three columns with
|
|
229
|
+
the verdicts drawn as lines between them, hover for the checker's
|
|
230
|
+
explanation. It opens with a double-click, needs no server and loads nothing
|
|
231
|
+
from the network. `--no-html` skips it.
|
|
232
|
+
`claimlens --help` lists all commands and flags.
|
|
233
|
+
|
|
234
|
+
## Use it from Python
|
|
235
|
+
|
|
236
|
+
The library is the same five verbs as the CLI, with the same keyword names
|
|
237
|
+
as its flags. Keys come from `EXTRACTOR_API_KEY` and `CHECKER_API_KEY` in
|
|
238
|
+
the environment or a `.env`, never as arguments.
|
|
239
|
+
|
|
240
|
+
```python
|
|
241
|
+
import json, claimlens
|
|
242
|
+
|
|
243
|
+
items = json.load(open("data.json"))
|
|
244
|
+
|
|
245
|
+
record, findings = claimlens.ragcheck(items, extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
|
|
246
|
+
record["metrics"]["faithfulness"]
|
|
247
|
+
findings["hallucination"] # the review queue, the same as the _findings.json file
|
|
248
|
+
|
|
249
|
+
items = claimlens.extract(items, extractor_model="gpt-4o-mini")
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
`ragcheck`, `faithcheck` and `refcheck` return the record and the findings,
|
|
253
|
+
the two documents the CLI writes. `claimlens.render_html(record, findings)`
|
|
254
|
+
returns the viewer page for a `ragcheck` record as a string, the same page
|
|
255
|
+
the CLI writes as `{report_stem}.html`. `extract` and `check` return the enriched
|
|
256
|
+
item list. Extra keyword arguments go to the pipeline (`concurrency`,
|
|
257
|
+
`joint`, `extractor_base_url`, `runs`, ...). Output is silent until you ask
|
|
258
|
+
for it: `claimlens.enable_logging()` turns the console output on, and
|
|
259
|
+
`verbosity="compact"` or `"silent"` per call sets how much.
|
|
260
|
+
|
|
261
|
+
**Inside a running event loop** (a Jupyter cell, an async server) the sync
|
|
262
|
+
verbs refuse with a message that says so. Use the pipeline classes with
|
|
263
|
+
`await`. It is the same code:
|
|
264
|
+
|
|
265
|
+
```python
|
|
266
|
+
from claimlens.pipelines.ragchecker import RagCheckerPipeline
|
|
267
|
+
|
|
268
|
+
pipeline = RagCheckerPipeline(extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
|
|
269
|
+
await pipeline.run(items)
|
|
270
|
+
record, findings = pipeline.last_report, pipeline.last_findings
|
|
271
|
+
```
|
|
272
|
+
|
|
273
|
+
**One response, in real time.** The async form first, because the caller
|
|
274
|
+
is usually a server:
|
|
275
|
+
|
|
276
|
+
```python
|
|
277
|
+
entry = await claimlens.acheck_faithfulness(response, chunks, extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
|
|
278
|
+
entry = claimlens.check_faithfulness(response, chunks, extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
|
|
279
|
+
```
|
|
280
|
+
|
|
281
|
+
The public API is what `claimlens/__init__.py` exports. Everything under
|
|
282
|
+
`claimlens.pipelines`, `.services` and `.workers` is importable but may
|
|
283
|
+
change between minor versions.
|
|
284
|
+
|
|
285
|
+
## Using other providers and local models
|
|
286
|
+
|
|
287
|
+
The extractor and checker are configured independently, and there are two ways
|
|
288
|
+
to reach a non-OpenAI endpoint.
|
|
289
|
+
|
|
290
|
+
**Direct endpoint** — pass a base URL with `--extractor-base-api` (or
|
|
291
|
+
`--checker-base-api`). The request goes straight out over the OpenAI SDK, and
|
|
292
|
+
the model name must **not** carry a provider prefix:
|
|
293
|
+
|
|
294
|
+
```bash
|
|
295
|
+
claimlens extract examples/extract/kepler22b.json \
|
|
296
|
+
--extractor-base-api https://openrouter.ai/api/v1 \
|
|
297
|
+
--extractor-model openai/gpt-5.6-luna
|
|
298
|
+
```
|
|
299
|
+
|
|
300
|
+
The same flag points at any OpenAI-compatible server, including a local one
|
|
301
|
+
(`--extractor-base-api http://localhost:11434/v1`).
|
|
302
|
+
|
|
303
|
+
**LiteLLM routing** (optional extra, `claimlens[litellm]`) — omit the base URL and prefix the model with its provider
|
|
304
|
+
instead. LiteLLM resolves the endpoint:
|
|
305
|
+
|
|
306
|
+
```bash
|
|
307
|
+
claimlens extract examples/extract/kepler22b.json --extractor-model openrouter/openai/gpt-5.6-luna
|
|
308
|
+
```
|
|
309
|
+
|
|
310
|
+
Prefer the direct endpoint for very new models: LiteLLM routing depends on the
|
|
311
|
+
installed LiteLLM release knowing the model, while a base URL bypasses that
|
|
312
|
+
lookup entirely. Either way the key comes from `EXTRACTOR_API_KEY` /
|
|
313
|
+
`CHECKER_API_KEY`.
|
|
314
|
+
|
|
315
|
+
## Documentation
|
|
316
|
+
|
|
317
|
+
- [examples/README.md](examples/README.md) — the five-step walkthrough:
|
|
318
|
+
what each command needs, what it produces, and a worked dataset with
|
|
319
|
+
deliberate errors planted in it
|
|
320
|
+
- [eval_data/msmarco/README.md](eval_data/msmarco/README.md) — provenance of
|
|
321
|
+
the human-labelled evaluation data, what was repaired in it, and what it can
|
|
322
|
+
and cannot measure
|
|
323
|
+
- [architecture.md](architecture.md) — the full design document (layer
|
|
324
|
+
model, data contract, error propagation)
|
|
325
|
+
- [docs/request_strategies.md](docs/request_strategies.md) — how requests
|
|
326
|
+
reach the model: capability discovery, retry behaviour, timeouts, and the
|
|
327
|
+
settings you can tune
|
|
328
|
+
- [docs/ragchecker.md](docs/ragchecker.md) — the RAG evaluation pipeline
|
|
329
|
+
- [docs/faithfulness.md](docs/faithfulness.md) — ground-truth-free
|
|
330
|
+
faithfulness checking
|
|
331
|
+
|
|
332
|
+
## Scientific lineage
|
|
333
|
+
|
|
334
|
+
The methodology follows RefChecker (Amazon Science) and RAGChecker
|
|
335
|
+
(claim-level RAG evaluation), with modernized components: explicit
|
|
336
|
+
abstention handling, a per-item error taxonomy instead of silent nulls,
|
|
337
|
+
claim deduplication, and variance reporting. It is an independent
|
|
338
|
+
reimplementation, not a fork.
|
|
339
|
+
|
|
340
|
+
## License
|
|
341
|
+
|
|
342
|
+
MIT
|