claimlens 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. claimlens-1.0.0/.env-example +4 -0
  2. claimlens-1.0.0/.gitattributes +8 -0
  3. claimlens-1.0.0/.gitignore +44 -0
  4. claimlens-1.0.0/.python-version +1 -0
  5. claimlens-1.0.0/.vscode/launch.json +224 -0
  6. claimlens-1.0.0/LICENSE +21 -0
  7. claimlens-1.0.0/PKG-INFO +342 -0
  8. claimlens-1.0.0/README.md +309 -0
  9. claimlens-1.0.0/architecture.md +269 -0
  10. claimlens-1.0.0/pyproject.toml +60 -0
  11. claimlens-1.0.0/src/claimlens/__init__.py +167 -0
  12. claimlens-1.0.0/src/claimlens/cli.py +715 -0
  13. claimlens-1.0.0/src/claimlens/eval/__init__.py +7 -0
  14. claimlens-1.0.0/src/claimlens/eval/base.py +111 -0
  15. claimlens-1.0.0/src/claimlens/eval/checkereval.py +693 -0
  16. claimlens-1.0.0/src/claimlens/eval/extractoreval.py +1326 -0
  17. claimlens-1.0.0/src/claimlens/eval/metrics.py +293 -0
  18. claimlens-1.0.0/src/claimlens/exceptions.py +93 -0
  19. claimlens-1.0.0/src/claimlens/llmclient.py +1068 -0
  20. claimlens-1.0.0/src/claimlens/models.py +192 -0
  21. claimlens-1.0.0/src/claimlens/pipelines/__init__.py +18 -0
  22. claimlens-1.0.0/src/claimlens/pipelines/directions.py +301 -0
  23. claimlens-1.0.0/src/claimlens/pipelines/faithfulness.py +620 -0
  24. claimlens-1.0.0/src/claimlens/pipelines/ragchecker.py +1096 -0
  25. claimlens-1.0.0/src/claimlens/pipelines/refchecker.py +363 -0
  26. claimlens-1.0.0/src/claimlens/prompt_map.json +12 -0
  27. claimlens-1.0.0/src/claimlens/services/atomization.py +441 -0
  28. claimlens-1.0.0/src/claimlens/services/base.py +244 -0
  29. claimlens-1.0.0/src/claimlens/services/checking.py +671 -0
  30. claimlens-1.0.0/src/claimlens/services/extraction.py +398 -0
  31. claimlens-1.0.0/src/claimlens/settings.py +161 -0
  32. claimlens-1.0.0/src/claimlens/stats.py +696 -0
  33. claimlens-1.0.0/src/claimlens/templates/common.css +52 -0
  34. claimlens-1.0.0/src/claimlens/templates/common.js +67 -0
  35. claimlens-1.0.0/src/claimlens/templates/ragcheck.html +289 -0
  36. claimlens-1.0.0/src/claimlens/utils.py +288 -0
  37. claimlens-1.0.0/src/claimlens/viewer.py +56 -0
  38. claimlens-1.0.0/src/claimlens/workers/atomizer.py +318 -0
  39. claimlens-1.0.0/src/claimlens/workers/checker.py +574 -0
  40. claimlens-1.0.0/src/claimlens/workers/extractor.py +288 -0
  41. claimlens-1.0.0/tests/conftest.py +18 -0
  42. claimlens-1.0.0/tests/integration/test_finish_reason_errors.py +256 -0
  43. claimlens-1.0.0/tests/integration/test_llmclient_fatal.py +484 -0
  44. claimlens-1.0.0/tests/integration/test_llmclient_retry.py +363 -0
  45. claimlens-1.0.0/tests/integration/test_llmclient_timeout.py +157 -0
  46. claimlens-1.0.0/tests/integration/test_timeout_end_to_end.py +106 -0
  47. claimlens-1.0.0/tests/unit/test_abstention.py +68 -0
  48. claimlens-1.0.0/tests/unit/test_atomization_service.py +355 -0
  49. claimlens-1.0.0/tests/unit/test_atomization_validate_filter.py +208 -0
  50. claimlens-1.0.0/tests/unit/test_atomizer_worker.py +287 -0
  51. claimlens-1.0.0/tests/unit/test_checkereval.py +595 -0
  52. claimlens-1.0.0/tests/unit/test_checking_service.py +731 -0
  53. claimlens-1.0.0/tests/unit/test_cli_output.py +117 -0
  54. claimlens-1.0.0/tests/unit/test_directions.py +244 -0
  55. claimlens-1.0.0/tests/unit/test_extraction_service.py +391 -0
  56. claimlens-1.0.0/tests/unit/test_extractoreval.py +968 -0
  57. claimlens-1.0.0/tests/unit/test_facades.py +138 -0
  58. claimlens-1.0.0/tests/unit/test_faithfulness.py +201 -0
  59. claimlens-1.0.0/tests/unit/test_llmclient_strategy_cache.py +119 -0
  60. claimlens-1.0.0/tests/unit/test_meta_request_strategies.py +163 -0
  61. claimlens-1.0.0/tests/unit/test_meta_usage.py +120 -0
  62. claimlens-1.0.0/tests/unit/test_metrics.py +182 -0
  63. claimlens-1.0.0/tests/unit/test_normalization.py +132 -0
  64. claimlens-1.0.0/tests/unit/test_output_conventions.py +399 -0
  65. claimlens-1.0.0/tests/unit/test_plain_prompts.py +171 -0
  66. claimlens-1.0.0/tests/unit/test_ragchecker.py +743 -0
  67. claimlens-1.0.0/tests/unit/test_refchecker_pipeline.py +276 -0
  68. claimlens-1.0.0/tests/unit/test_report_envelope.py +161 -0
  69. claimlens-1.0.0/tests/unit/test_runs.py +195 -0
  70. claimlens-1.0.0/tests/unit/test_schema_shape.py +124 -0
  71. claimlens-1.0.0/tests/unit/test_timeout_downstream.py +112 -0
  72. claimlens-1.0.0/tests/unit/test_utils.py +197 -0
  73. claimlens-1.0.0/tests/unit/test_viewer.py +163 -0
  74. claimlens-1.0.0/uv.lock +1734 -0
@@ -0,0 +1,4 @@
1
+ EXTRACTOR_API_KEY="sk-1234"
2
+ CHECKER_API_KEY="sk-1234"
3
+ LLM_TIMEOUT=120 # not sure this is a seceret?
4
+ # LLM_MAX_TOKENS=500 # unset = no cap; set it to cut answers short on purpose
@@ -0,0 +1,8 @@
1
+ # Normalize line endings: source stays LF in the repo AND in the working
2
+ # tree, so Windows editors stop flip-flopping and git stops warning
3
+ # "LF will be replaced by CRLF" on every command.
4
+ * text=auto
5
+ *.py text eol=lf
6
+ *.md text eol=lf
7
+ *.toml text eol=lf
8
+ *.json text eol=lf
@@ -0,0 +1,44 @@
1
+ # ── Build / Dist ─────────────────────────────────────────────────────────────
2
+ dist/
3
+ build/
4
+ *.egg-info/
5
+
6
+ # ── Python ───────────────────────────────────────────────────────────────────
7
+ __pycache__/
8
+ *.py[cod]
9
+ *.pyo
10
+
11
+ # ── Virtual Environments ────────────────────────────────────────────────────
12
+ .venv/
13
+ venv/
14
+
15
+ # ── IDE ──────────────────────────────────────────────────────────────────────
16
+ # launch.json is shared: it documents the standard workflows. Everything else
17
+ # under .vscode/ is personal.
18
+ .vscode/*
19
+ !.vscode/launch.json
20
+ .idea/
21
+
22
+ # ── Test / Coverage ─────────────────────────────────────────────────────────
23
+ .pytest_cache/
24
+ .coverage
25
+ htmlcov/
26
+
27
+ # ── Secrets / Env ────────────────────────────────────────────────────────────
28
+ *.env
29
+ .env*
30
+ !.env-example
31
+
32
+ # ── OS ───────────────────────────────────────────────────────────────────────
33
+ .DS_Store
34
+ Thumbs.db
35
+
36
+ # ── Results (local runs) ────────────────────────────────────────────────────
37
+ # Outputs land next to their input, so examples/ and eval_data/ grow their own.
38
+ **/results/
39
+
40
+ ignore-found-bugs.md
41
+
42
+ agent.md
43
+
44
+ eval_cases_reference.md
@@ -0,0 +1 @@
1
+ 3.12
@@ -0,0 +1,224 @@
1
+ {
2
+ "version": "0.2.0",
3
+ "inputs": [
4
+ {
5
+ "id": "extractorModel",
6
+ "type": "promptString",
7
+ "description": "Extractor model — bare id when a base URL is set (e.g. llama3.2:3b, openai/gpt-5.6-luna)",
8
+ "default": "z-ai/glm-5.3-flash"
9
+ },
10
+ {
11
+ "id": "extractorBaseApi",
12
+ "type": "promptString",
13
+ "description": "Extractor base URL — leave blank for LiteLLM provider routing (then prefix the model, e.g. openrouter/...)",
14
+ "default": "https://openrouter.ai/api/v1"
15
+ },
16
+ {
17
+ "id": "checkerModel",
18
+ "type": "promptString",
19
+ "description": "Checker model — bare id when a base URL is set (e.g. llama3.2:3b, openai/gpt-5.6-luna)",
20
+ "default": "z-ai/glm-5.3-flash"
21
+ },
22
+ {
23
+ "id": "checkerBaseApi",
24
+ "type": "promptString",
25
+ "description": "Checker base URL — leave blank for LiteLLM provider routing (then prefix the model, e.g. openrouter/...)",
26
+ "default": "https://openrouter.ai/api/v1"
27
+ },
28
+ {
29
+ "id": "atomizerModel",
30
+ "type": "promptString",
31
+ "description": "Atomizer model — measures whether extracted claims are atomic. Needs ATOMIZER_API_KEY.",
32
+ "default": "z-ai/glm-5.3-flash"
33
+ },
34
+ {
35
+ "id": "atomizerBaseApi",
36
+ "type": "promptString",
37
+ "description": "Atomizer base URL — leave blank for LiteLLM provider routing",
38
+ "default": "https://openrouter.ai/api/v1"
39
+ }
40
+ ],
41
+ "configurations": [
42
+ {
43
+ "name": "01 · extract — kepler22b",
44
+ "type": "debugpy",
45
+ "request": "launch",
46
+ "program": "${workspaceFolder}/.venv/bin/claimlens",
47
+ "args": [
48
+ "extract",
49
+ "examples/extract/kepler22b.json",
50
+ "--extractor-model",
51
+ "${input:extractorModel}",
52
+ "--extractor-base-api",
53
+ "${input:extractorBaseApi}"
54
+ ],
55
+ "python": "${workspaceFolder}/.venv/bin/python",
56
+ "cwd": "${workspaceFolder}",
57
+ "console": "integratedTerminal",
58
+ "justMyCode": false,
59
+ "presentation": {
60
+ "group": "1 · Walkthrough",
61
+ "order": 1
62
+ }
63
+ },
64
+ {
65
+ "name": "02 · check — kepler22b (extracted)",
66
+ "type": "debugpy",
67
+ "request": "launch",
68
+ "program": "${workspaceFolder}/.venv/bin/claimlens",
69
+ "args": [
70
+ "check",
71
+ "examples/check/kepler22b_extracted.json",
72
+ "--extractor-model",
73
+ "openai/gpt-5.6-luna",
74
+ "--checker-model",
75
+ "${input:checkerModel}",
76
+ "--checker-base-api",
77
+ "${input:checkerBaseApi}"
78
+ ],
79
+ "python": "${workspaceFolder}/.venv/bin/python",
80
+ "cwd": "${workspaceFolder}",
81
+ "console": "integratedTerminal",
82
+ "justMyCode": false,
83
+ "presentation": {
84
+ "group": "1 · Walkthrough",
85
+ "order": 2
86
+ }
87
+ },
88
+ {
89
+ "name": "03 · refcheck — kepler22b",
90
+ "type": "debugpy",
91
+ "request": "launch",
92
+ "program": "${workspaceFolder}/.venv/bin/claimlens",
93
+ "args": [
94
+ "refcheck",
95
+ "examples/refcheck/kepler22b.json",
96
+ "--extractor-model",
97
+ "${input:extractorModel}",
98
+ "--extractor-base-api",
99
+ "${input:extractorBaseApi}",
100
+ "--checker-model",
101
+ "${input:checkerModel}",
102
+ "--checker-base-api",
103
+ "${input:checkerBaseApi}",
104
+ "--debug"
105
+ ],
106
+ "python": "${workspaceFolder}/.venv/bin/python",
107
+ "cwd": "${workspaceFolder}",
108
+ "console": "integratedTerminal",
109
+ "justMyCode": false,
110
+ "presentation": {
111
+ "group": "1 · Walkthrough",
112
+ "order": 3
113
+ }
114
+ },
115
+ {
116
+ "name": "04 · faithcheck — kepler22b",
117
+ "type": "debugpy",
118
+ "request": "launch",
119
+ "program": "${workspaceFolder}/.venv/bin/claimlens",
120
+ "args": [
121
+ "faithcheck",
122
+ "examples/faithcheck/kepler22b.json",
123
+ "--extractor-model",
124
+ "${input:extractorModel}",
125
+ "--extractor-base-api",
126
+ "${input:extractorBaseApi}",
127
+ "--checker-model",
128
+ "${input:checkerModel}",
129
+ "--checker-base-api",
130
+ "${input:checkerBaseApi}",
131
+ "--runs",
132
+ "2"
133
+ ],
134
+ "python": "${workspaceFolder}/.venv/bin/python",
135
+ "cwd": "${workspaceFolder}",
136
+ "console": "integratedTerminal",
137
+ "justMyCode": false,
138
+ "presentation": {
139
+ "group": "1 · Walkthrough",
140
+ "order": 4
141
+ }
142
+ },
143
+ {
144
+ "name": "05 · ragcheck — kepler22b",
145
+ "type": "debugpy",
146
+ "request": "launch",
147
+ "program": "${workspaceFolder}/.venv/bin/claimlens",
148
+ "args": [
149
+ "ragcheck",
150
+ "examples/ragcheck/kepler22b.json",
151
+ "--extractor-model",
152
+ "${input:extractorModel}",
153
+ "--extractor-base-api",
154
+ "${input:extractorBaseApi}",
155
+ "--checker-model",
156
+ "${input:checkerModel}",
157
+ "--checker-base-api",
158
+ "${input:checkerBaseApi}",
159
+ "--runs",
160
+ "2"
161
+ ],
162
+ "python": "${workspaceFolder}/.venv/bin/python",
163
+ "cwd": "${workspaceFolder}",
164
+ "console": "integratedTerminal",
165
+ "justMyCode": false,
166
+ "presentation": {
167
+ "group": "1 · Walkthrough",
168
+ "order": 5
169
+ }
170
+ },
171
+ {
172
+ "type": "debugpy",
173
+ "request": "launch",
174
+ "program": "${workspaceFolder}/.venv/bin/claimlens",
175
+ "python": "${workspaceFolder}/.venv/bin/python",
176
+ "cwd": "${workspaceFolder}",
177
+ "console": "integratedTerminal",
178
+ "justMyCode": false,
179
+ "name": "01 · eval checker — msmarco_gpt4_5‚",
180
+ "args": [
181
+ "eval",
182
+ "checker",
183
+ "eval_data/msmarco/msmarco_gpt4_25.json",
184
+ "--checker-model",
185
+ "${input:checkerModel}",
186
+ "--checker-base-api",
187
+ "${input:checkerBaseApi}",
188
+ "--runs",
189
+ "1"
190
+ ],
191
+ "presentation": {
192
+ "group": "2 · Evaluation",
193
+ "order": 1
194
+ }
195
+ },
196
+ {
197
+ "type": "debugpy",
198
+ "request": "launch",
199
+ "program": "${workspaceFolder}/.venv/bin/claimlens",
200
+ "python": "${workspaceFolder}/.venv/bin/python",
201
+ "cwd": "${workspaceFolder}",
202
+ "console": "integratedTerminal",
203
+ "justMyCode": false,
204
+ "name": "02 · eval extractor — msmarco_gpt4_25",
205
+ "args": [
206
+ "eval",
207
+ "extractor",
208
+ "eval_data/msmarco/msmarco_gpt4_5.json",
209
+ "--extractor-model",
210
+ "${input:extractorModel}",
211
+ "--extractor-base-api",
212
+ "${input:extractorBaseApi}",
213
+ "--checker-model",
214
+ "${input:checkerModel}",
215
+ "--checker-base-api",
216
+ "${input:checkerBaseApi}"
217
+ ],
218
+ "presentation": {
219
+ "group": "2 · Evaluation",
220
+ "order": 2
221
+ }
222
+ }
223
+ ]
224
+ }
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Arthur Galas
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,342 @@
1
+ Metadata-Version: 2.5
2
+ Name: claimlens
3
+ Version: 1.0.0
4
+ Summary: Claim-level evaluation for LLM outputs: decompose text into atomic claims, then verify every claim against a reference.
5
+ Project-URL: Homepage, https://github.com/Arthur-g-p/claimlens
6
+ Project-URL: Repository, https://github.com/Arthur-g-p/claimlens
7
+ Project-URL: Issues, https://github.com/Arthur-g-p/claimlens/issues
8
+ Author: Arthur Galas
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: claim-verification,evaluation,faithfulness,hallucination,llm,rag
12
+ Classifier: Development Status :: 5 - Production/Stable
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Requires-Python: >=3.12
19
+ Requires-Dist: httpx>=0.27.0
20
+ Requires-Dist: openai>=1.0
21
+ Requires-Dist: pydantic>=2.0
22
+ Requires-Dist: python-dotenv>=1.0.0
23
+ Requires-Dist: tqdm>=4.60
24
+ Requires-Dist: typer>=0.9.0
25
+ Provides-Extra: litellm
26
+ Requires-Dist: litellm>=1.0.0; extra == 'litellm'
27
+ Provides-Extra: test
28
+ Requires-Dist: coverage>=7.0; extra == 'test'
29
+ Requires-Dist: litellm>=1.0.0; extra == 'test'
30
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'test'
31
+ Requires-Dist: pytest>=8.0; extra == 'test'
32
+ Description-Content-Type: text/markdown
33
+
34
+ # ClaimLens
35
+
36
+ Claim-level evaluation for LLM outputs: decompose text into atomic claims,
37
+ then verify every claim against a reference.
38
+
39
+ Instead of asking a judge LLM "rate this answer 1-10", ClaimLens
40
+ extracts atomic `(subject, predicate, object)` claims from a response with
41
+ an extractor LLM, then classifies each claim against a reference with a
42
+ checker LLM: Entailment, Contradiction, or Neutral. Every score the toolkit
43
+ produces is an aggregate over per-claim verdicts you can read, audit, and
44
+ disagree with.
45
+
46
+ ## Commands
47
+
48
+ | Command | What it does |
49
+ | --- | --- |
50
+ | `ragcheck` | RAGChecker-style RAG evaluation: 2 extractions + 4 checking directions, one self-contained JSON report (precision, recall, F1, faithfulness, and claim-level detail) |
51
+ | `faithcheck` | Faithfulness checking **without ground truth**: response claims vs the retrieved context. Works on live traffic. |
52
+ | `refcheck` | Classic reference checking: extraction + checking in one pass |
53
+ | `extract` / `check` / `atomize` | The individual building blocks, runnable standalone — walked through in order under [examples/](examples/README.md) |
54
+ | `eval extractor` / `eval checker` | Meta-evaluation: measure the extractor and checker themselves against labeled data. Run `eval checker` first — `eval extractor` uses a checker, so an unqualified one makes its numbers partial |
55
+
56
+ ## Why this instead of a single judge score
57
+
58
+ - **Auditable**: every metric decomposes into claim-level verdicts with
59
+ explanations. You can see exactly which fact failed and why.
60
+ - **Variance is first-class**: LLM-based metrics are stochastic. Pass
61
+ `--runs N` to `ragcheck`, `faithcheck`, `eval extractor` or
62
+ `eval checker` and get `mean +/- std [min, max]` per metric across N
63
+ full repetitions, plus each run's complete report. A single-run score
64
+ overstates your certainty.
65
+ - **The evaluator is itself evaluated**: `eval extractor` and
66
+ `eval checker` measure the measurement tool against ground-truth
67
+ labels, so you know how much to trust the numbers before you compare
68
+ systems with them.
69
+ - **Honest failure handling**: per-item errors (context too long, content
70
+ policy, parse failure) are recorded as values next to the affected claim,
71
+ never silently zeroed. A claim the checker failed to judge leaves both
72
+ sides of the ratio rather than counting against the system under test,
73
+ and a metric with an empty denominator is `null`, never `0.0`.
74
+ Abstentions ("I don't know") are first-class labeled outcomes: judged
75
+ justified or unjustified, their cause apportioned between retriever
76
+ and generator, and unanswerable questions (`"gt_answer": ""`) expose
77
+ unwarranted answers — while the metrics stay standard (a refusal is
78
+ non-delivery; see docs/ragchecker.md#abstention).
79
+ - **Any OpenAI-compatible endpoint**: point the extractor and checker at
80
+ different models, providers, or local servers independently.
81
+
82
+ ## Installation
83
+
84
+ Requires Python 3.12+. The project uses [uv](https://docs.astral.sh/uv/) for
85
+ dependency management — install it first if you don't have it:
86
+
87
+ ```bash
88
+ # macOS / Linux
89
+ curl -LsSf https://astral.sh/uv/install.sh | sh
90
+ ```
91
+
92
+ ```bash
93
+ # macOS (Homebrew alternative)
94
+ brew install uv
95
+ ```
96
+
97
+ ```powershell
98
+ # Windows (PowerShell)
99
+ powershell -ExecutionPolicy ByPass -c "irm https://astral.sh/uv/install.ps1 | iex"
100
+ ```
101
+
102
+ Then clone and install into a virtual environment:
103
+
104
+ ```bash
105
+ git clone https://github.com/Arthur-g-p/claimlens.git
106
+ cd claimlens
107
+ uv venv
108
+ uv pip install -e .
109
+ ```
110
+
111
+ `uv venv` reads `.python-version` and downloads CPython 3.12 if it isn't
112
+ already present. It does not touch your system Python.
113
+
114
+ Activate the environment before using the `claimlens` command:
115
+
116
+ ```bash
117
+ source .venv/bin/activate # macOS / Linux
118
+ ```
119
+
120
+ ```powershell
121
+ .venv\Scripts\activate # Windows
122
+ ```
123
+
124
+ Activation is per-shell — you need it again in every new terminal. To skip it,
125
+ prefix commands with `uv run` instead (`uv run claimlens --help`).
126
+
127
+ To also install the test dependencies (pytest, pytest-asyncio, coverage):
128
+
129
+ ```bash
130
+ uv pip install -e ".[test]"
131
+ ```
132
+
133
+ LiteLLM routing (a provider prefix on the model name, no base URL) is an
134
+ optional extra. The default install talks to any OpenAI-compatible endpoint
135
+ directly and stays small. To add the routing path:
136
+
137
+ ```bash
138
+ uv pip install -e ".[litellm]"
139
+ ```
140
+
141
+ From PyPI that is `pip install "claimlens[litellm]"`.
142
+
143
+ Verify the install:
144
+
145
+ ```bash
146
+ claimlens --help
147
+ ```
148
+
149
+ ## Quick start
150
+
151
+ Copy the environment template and fill in your API keys:
152
+
153
+ ```bash
154
+ cp .env-example .env
155
+ ```
156
+
157
+ ```
158
+ EXTRACTOR_API_KEY=...
159
+ CHECKER_API_KEY=...
160
+ ```
161
+
162
+ Input is a JSON list of items. For `ragcheck`, each item needs `response`,
163
+ `gt_answer`, and `retrieved_context`:
164
+
165
+ ```json
166
+ [
167
+ {
168
+ "question": "Who wrote Faust?",
169
+ "response": "Faust was written by Goethe in the 19th century.",
170
+ "gt_answer": "Johann Wolfgang von Goethe wrote Faust.",
171
+ "retrieved_context": ["Goethe published Faust, Part One in 1808. ..."]
172
+ }
173
+ ]
174
+ ```
175
+
176
+ The input file is a **positional** argument on every command — pass the path
177
+ bare, with no flag in front of it.
178
+
179
+ You do not need to prepare data to start. `examples/` ships a five-step
180
+ walkthrough over one dataset, where each step needs a little more than the last,
181
+ plus `eval_data/` with human-labelled data for the `eval` commands.
182
+
183
+ Start at step 01 — decompose responses into claims:
184
+
185
+ ```bash
186
+ claimlens extract examples/extract/kepler22b.json --extractor-model gpt-4o-mini
187
+ ```
188
+
189
+ Run the full RAG pipeline (step 05), which needs `gt_answer` and
190
+ `retrieved_context` as well:
191
+
192
+ ```bash
193
+ claimlens ragcheck examples/ragcheck/kepler22b.json --extractor-model gpt-4o-mini --checker-model gpt-4o-mini
194
+ ```
195
+
196
+ Repeat the whole experiment five times and report variance (5x LLM cost):
197
+
198
+ ```bash
199
+ claimlens ragcheck examples/ragcheck/kepler22b.json --extractor-model gpt-4o-mini --checker-model gpt-4o-mini --runs 5
200
+ ```
201
+
202
+ Check faithfulness with no ground truth at all (step 04):
203
+
204
+ ```bash
205
+ claimlens faithcheck examples/faithcheck/kepler22b.json --extractor-model gpt-4o-mini --checker-model gpt-4o-mini
206
+ ```
207
+
208
+ The model flags have short aliases (`-e`, `-c`, `-m`), but the long names are
209
+ spelled the same way in every command, so those are the ones worth learning.
210
+ [examples/README.md](examples/README.md) lists which fields and which arguments
211
+ each step needs.
212
+
213
+ The pipelines and evals (`ragcheck`, `faithcheck`, `refcheck`, `eval
214
+ extractor`, `eval checker`) write two JSON files. The record holds everything: an `_args` block (what
215
+ you asked for — every flag of the invocation), a `_meta` block (what the run
216
+ turned out to be — counts, timings, derived keys), `metrics` where the command
217
+ computes any, every count the console prints, and the complete per-item
218
+ claim record — `--runs N` adds entries, it never reshapes. The
219
+ `_findings.json` sibling is the review queue: only what went wrong, keyed by
220
+ the console's own branch names, derived from the record. The building
221
+ blocks (`extract`, `check`, `atomize`) instead emit the item list itself,
222
+ enriched in place, so each one's output is the next one's input.
223
+
224
+ `ragcheck` writes a third file next to those two: `{report_stem}.html`, a
225
+ self-contained viewer of the same record. One screen for the run — the main
226
+ metrics with their spread over `--runs`, then every item as a row of dots,
227
+ one per response claim, lime when a chunk grounds it — and one screen per
228
+ item: GT claims, retrieved chunks and response claims in three columns with
229
+ the verdicts drawn as lines between them, hover for the checker's
230
+ explanation. It opens with a double-click, needs no server and loads nothing
231
+ from the network. `--no-html` skips it.
232
+ `claimlens --help` lists all commands and flags.
233
+
234
+ ## Use it from Python
235
+
236
+ The library is the same five verbs as the CLI, with the same keyword names
237
+ as its flags. Keys come from `EXTRACTOR_API_KEY` and `CHECKER_API_KEY` in
238
+ the environment or a `.env`, never as arguments.
239
+
240
+ ```python
241
+ import json, claimlens
242
+
243
+ items = json.load(open("data.json"))
244
+
245
+ record, findings = claimlens.ragcheck(items, extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
246
+ record["metrics"]["faithfulness"]
247
+ findings["hallucination"] # the review queue, the same as the _findings.json file
248
+
249
+ items = claimlens.extract(items, extractor_model="gpt-4o-mini")
250
+ ```
251
+
252
+ `ragcheck`, `faithcheck` and `refcheck` return the record and the findings,
253
+ the two documents the CLI writes. `claimlens.render_html(record, findings)`
254
+ returns the viewer page for a `ragcheck` record as a string, the same page
255
+ the CLI writes as `{report_stem}.html`. `extract` and `check` return the enriched
256
+ item list. Extra keyword arguments go to the pipeline (`concurrency`,
257
+ `joint`, `extractor_base_url`, `runs`, ...). Output is silent until you ask
258
+ for it: `claimlens.enable_logging()` turns the console output on, and
259
+ `verbosity="compact"` or `"silent"` per call sets how much.
260
+
261
+ **Inside a running event loop** (a Jupyter cell, an async server) the sync
262
+ verbs refuse with a message that says so. Use the pipeline classes with
263
+ `await`. It is the same code:
264
+
265
+ ```python
266
+ from claimlens.pipelines.ragchecker import RagCheckerPipeline
267
+
268
+ pipeline = RagCheckerPipeline(extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
269
+ await pipeline.run(items)
270
+ record, findings = pipeline.last_report, pipeline.last_findings
271
+ ```
272
+
273
+ **One response, in real time.** The async form first, because the caller
274
+ is usually a server:
275
+
276
+ ```python
277
+ entry = await claimlens.acheck_faithfulness(response, chunks, extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
278
+ entry = claimlens.check_faithfulness(response, chunks, extractor_model="gpt-4o-mini", checker_model="gpt-4o-mini")
279
+ ```
280
+
281
+ The public API is what `claimlens/__init__.py` exports. Everything under
282
+ `claimlens.pipelines`, `.services` and `.workers` is importable but may
283
+ change between minor versions.
284
+
285
+ ## Using other providers and local models
286
+
287
+ The extractor and checker are configured independently, and there are two ways
288
+ to reach a non-OpenAI endpoint.
289
+
290
+ **Direct endpoint** — pass a base URL with `--extractor-base-api` (or
291
+ `--checker-base-api`). The request goes straight out over the OpenAI SDK, and
292
+ the model name must **not** carry a provider prefix:
293
+
294
+ ```bash
295
+ claimlens extract examples/extract/kepler22b.json \
296
+ --extractor-base-api https://openrouter.ai/api/v1 \
297
+ --extractor-model openai/gpt-5.6-luna
298
+ ```
299
+
300
+ The same flag points at any OpenAI-compatible server, including a local one
301
+ (`--extractor-base-api http://localhost:11434/v1`).
302
+
303
+ **LiteLLM routing** (optional extra, `claimlens[litellm]`) — omit the base URL and prefix the model with its provider
304
+ instead. LiteLLM resolves the endpoint:
305
+
306
+ ```bash
307
+ claimlens extract examples/extract/kepler22b.json --extractor-model openrouter/openai/gpt-5.6-luna
308
+ ```
309
+
310
+ Prefer the direct endpoint for very new models: LiteLLM routing depends on the
311
+ installed LiteLLM release knowing the model, while a base URL bypasses that
312
+ lookup entirely. Either way the key comes from `EXTRACTOR_API_KEY` /
313
+ `CHECKER_API_KEY`.
314
+
315
+ ## Documentation
316
+
317
+ - [examples/README.md](examples/README.md) — the five-step walkthrough:
318
+ what each command needs, what it produces, and a worked dataset with
319
+ deliberate errors planted in it
320
+ - [eval_data/msmarco/README.md](eval_data/msmarco/README.md) — provenance of
321
+ the human-labelled evaluation data, what was repaired in it, and what it can
322
+ and cannot measure
323
+ - [architecture.md](architecture.md) — the full design document (layer
324
+ model, data contract, error propagation)
325
+ - [docs/request_strategies.md](docs/request_strategies.md) — how requests
326
+ reach the model: capability discovery, retry behaviour, timeouts, and the
327
+ settings you can tune
328
+ - [docs/ragchecker.md](docs/ragchecker.md) — the RAG evaluation pipeline
329
+ - [docs/faithfulness.md](docs/faithfulness.md) — ground-truth-free
330
+ faithfulness checking
331
+
332
+ ## Scientific lineage
333
+
334
+ The methodology follows RefChecker (Amazon Science) and RAGChecker
335
+ (claim-level RAG evaluation), with modernized components: explicit
336
+ abstention handling, a per-item error taxonomy instead of silent nulls,
337
+ claim deduplication, and variance reporting. It is an independent
338
+ reimplementation, not a fork.
339
+
340
+ ## License
341
+
342
+ MIT