rag-quality-check 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ # Virtual environments and scratch space used while building. Never committed:
2
+ # they are large, machine-specific, and rebuilt from pyproject.toml anyway.
3
+ _venvs/
4
+ _proof/
5
+ .venv*/
6
+
7
+ # Build output. Wheels are built by the release workflow from this source, so a
8
+ # wheel in git could silently differ from the code beside it.
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ src/*.egg-info/
13
+
14
+ # Python noise
15
+ __pycache__/
16
+ *.py[cod]
17
+ .pytest_cache/
18
+ .mypy_cache/
19
+ .ruff_cache/
20
+
21
+ # Local verification state, not source
22
+ verify_results.json
23
+ publish-log.txt
24
+ published.json
25
+
26
+ # Credentials. None of these belong here, and this line is the backstop.
27
+ .pypirc
28
+ .env
29
+ *.pem
30
+ *.key
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Pranay Mahendrakar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,202 @@
1
+ Metadata-Version: 2.5
2
+ Name: rag-quality-check
3
+ Version: 0.1.0
4
+ Summary: Measure whether a retrieval system is actually retrieving the right things
5
+ Project-URL: Homepage, https://pypi.org/project/rag-quality-check/
6
+ Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
7
+ Author: Pranay Mahendrakar
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: evaluation,groundedness,information-retrieval,llm,mrr,ndcg,rag,retrieval
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Topic :: Text Processing :: Linguistic
19
+ Requires-Python: >=3.9
20
+ Requires-Dist: numpy>=1.23
21
+ Provides-Extra: dev
22
+ Requires-Dist: pandas>=1.3; extra == 'dev'
23
+ Requires-Dist: pytest>=7; extra == 'dev'
24
+ Provides-Extra: frame
25
+ Requires-Dist: pandas>=1.3; extra == 'frame'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # rag-quality-check
29
+
30
+ Your RAG answers are only as good as what the retriever handed the model, and
31
+ "it feels better now" is not a number - this measures whether the right passages
32
+ actually came back, and whether the answer stayed inside them.
33
+
34
+ ## Install
35
+
36
+ ```bash
37
+ pip install rag-quality-check
38
+ ```
39
+
40
+ ## Quickstart
41
+
42
+ ```python
43
+ import rag_quality_check
44
+
45
+ cases = [
46
+ {"query": "reset my password", "retrieved": ["p_reset", "p_billing"], "relevant": ["p_reset"]},
47
+ {"query": "cancel my plan", "retrieved": ["p_faq", "p_billing"], "relevant": ["p_cancel"]},
48
+ ]
49
+ print(rag_quality_check.evaluate(cases, k=2).summary())
50
+ ```
51
+
52
+ ```
53
+ rag-quality-check: 2 case(s), top-2, threshold 0.5, similarity by lexical overlap.
54
+ retrieval:
55
+ precision_at_k 0.250 relevant items in the top k, divided by k (2 case(s))
56
+ recall_at_k 0.500 relevant items found, divided by all relevant ones (2 case(s))
57
+ mrr 0.500 1 / rank of the first relevant item, averaged over cases (2 case(s))
58
+ ndcg_at_k 0.500 DCG@k / ideal DCG@k, linear gains, log2 discount (2 case(s))
59
+ hit_rate 0.500 share of queries with a relevant item in the top k (2 case(s))
60
+ failures (1):
61
+ - case 1 'cancel my plan': nothing relevant in the top 2
62
+ weakest cases:
63
+ [1] 0.00 'cancel my plan'
64
+ [0] 0.90 'reset my password'
65
+ ```
66
+
67
+ ## What it measures
68
+
69
+ - **precision_at_k**, **recall_at_k**, **mrr**, **ndcg_at_k**, **hit_rate** - the
70
+ five retrieval numbers, each with its exact formula written out in its
71
+ docstring. Users compare these across systems, so an undocumented variant of
72
+ nDCG is worse than no nDCG at all.
73
+ - **groundedness** - the share of the answer's words that a retrieved passage
74
+ actually supports. This is the hallucination check. Sentences are weighted by
75
+ how many content words they hold, so a leading `"Yes."` cannot halve the score
76
+ of a paragraph copied straight out of the context.
77
+ - **citation_coverage** - the share of retrieved passages the answer drew on. A
78
+ low number beside a high groundedness means the retriever is returning padding.
79
+ - **answer_relevance** - how much of the query's own wording the answer repeats,
80
+ and **answer_correctness** when you pass a `ground_truth`.
81
+ - Every metric is in `0..1`. Every one of them is defined once, in one place.
82
+
83
+ It also tells you what it could *not* measure:
84
+
85
+ - No `relevant` ids on a case? Relevance is **estimated** from similarity to the
86
+ query, and the report says "ESTIMATED" every time it prints, on the report and
87
+ on the case. Estimated numbers are for spotting weak queries, not for
88
+ comparing two systems.
89
+ - A case whose `relevant` list is empty is a query you have labelled as having
90
+ no right answer. Every retrieval metric is **excluded** for it and the
91
+ exclusion is noted: retrieving nothing relevant is the wanted result there, so
92
+ scoring it `0.0` would drag the means down and fill `failures` with successes.
93
+ - Passed ids in `retrieved` instead of passage text? The retrieval metrics are
94
+ unaffected, but `groundedness` and `citation_coverage` are **excluded** with a
95
+ note rather than reported as `0.0` - an id carries no words for an answer to
96
+ be supported by, and scoring against one would report a hallucination that
97
+ never happened. Pass `{"id": ..., "text": ...}` to measure them.
98
+ - `answer_relevance` under the default word overlap is the share of the query's
99
+ content words the answer repeats, and a correct answer routinely repeats none
100
+ of them ("how much does the pro plan cost" answered by "It is 29 dollars per
101
+ seat per month." scores `0.00`). It is reported, noted when it is low, and
102
+ never counted as a failure unless you pass `embed=`, where it is a cosine
103
+ between meanings and does mean what it says.
104
+ - If no `relevant` id matches any retrieved id anywhere in the run, the report
105
+ says so, because ids compared in two different forms and a broken retriever
106
+ produce the same zeros.
107
+ - An empty `retrieved` list scores zeros, never a `ZeroDivisionError`.
108
+ - Asked for top-5 and only 3 came back? `precision_at_k` still divides by 5, and
109
+ the case says so in a note instead of silently shrinking `k`.
110
+ - Duplicate passages in `retrieved` are counted once, and the duplicate count is
111
+ noted.
112
+ - Unicode queries and passages work; CJK text is compared character by
113
+ character, since it is not written with spaces. One caveat, and only in
114
+ estimated mode: word overlap matches word forms, so an inflected language can
115
+ miss a passage that plainly answers the query (Hindi "बदलें" against "बदलने"
116
+ counts as a miss). Labelled `relevant` ids are unaffected, and `embed=` fixes
117
+ the estimate.
118
+ - 1000 cases score in about a second. No model downloads, no network, numpy only.
119
+
120
+ By default similarity is word overlap. Pass `embed=` any
121
+ `callable(list[str]) -> vectors` - a sentence-transformer `.encode`, an API
122
+ client, anything - and every similarity becomes cosine instead, so you get real
123
+ semantics without this package depending on a model:
124
+
125
+ ```python
126
+ report = rag_quality_check.evaluate(cases, embed=model.encode)
127
+ ```
128
+
129
+ Texts are de-duplicated and embedded in one batched call for the whole run.
130
+
131
+ ## API
132
+
133
+ ### `evaluate(cases, *, k=5, threshold=0.5, embed=None) -> RagReport`
134
+
135
+ `cases` is a list of dicts:
136
+
137
+ | key | required | meaning |
138
+ | --- | --- | --- |
139
+ | `query` | yes | the question that was asked |
140
+ | `retrieved` | yes | what came back, best first: passage strings, ids, or dicts with `id` and/or `text`. Ids alone are enough for the retrieval metrics; `groundedness` and `citation_coverage` need the text |
141
+ | `relevant` | no | ids that should have come back - a list, or `{id: gain}` for graded relevance. Without it, relevance is estimated; an empty list means the query has no right answer |
142
+ | `answer` | no | the generated answer; turns on the answer metrics |
143
+ | `ground_truth` | no | the reference answer; adds `answer_correctness` |
144
+
145
+ `k` is the cut-off and is always the denominator of `precision_at_k`.
146
+ `threshold` is the similarity at which two texts count as a match, and the score
147
+ below which a case is reported as a failure - with the single exception of
148
+ `answer_relevance` under word overlap, which is reported and never accused.
149
+
150
+ ### `evaluate_case(query, retrieved, **kw) -> CaseResult`
151
+
152
+ One query, same rules and the same numbers as a one-case `evaluate`. Takes
153
+ `relevant`, `answer`, `ground_truth`, `k`, `threshold` and `embed`.
154
+
155
+ ```python
156
+ case = rag_quality_check.evaluate_case(
157
+ "how do I reset my password",
158
+ ["To reset your password, open Settings and choose Reset."],
159
+ answer="Open Settings and choose Reset.",
160
+ k=1,
161
+ )
162
+ print(case.summary())
163
+ ```
164
+
165
+ ### `RagReport`
166
+
167
+ | member | what it is |
168
+ | --- | --- |
169
+ | `.metrics` | `dict[str, float]`, each metric averaged over the cases where it was defined |
170
+ | `.support` | how many cases went into each mean |
171
+ | `.per_case` | `list[CaseResult]` |
172
+ | `.weakest(n=5)` | the `n` worst cases, weakest first |
173
+ | `.failures` | `list[str]`, plain language: retrieved nothing, nothing relevant in the top k, answer not grounded, ... |
174
+ | `.notes` | everything that changed how a number was produced |
175
+ | `.estimated` / `.fully_estimated` | whether any / every case's relevance was guessed |
176
+ | `.similarity` | `"lexical overlap"` or `"embeddings"` |
177
+ | `.ok` | `True` when no case failed |
178
+ | `.summary()` | the human report above |
179
+ | `.to_dict()` | JSON-safe, per-case detail included |
180
+ | `.to_frame()` | one row per case as a `pandas.DataFrame` (needs `pip install rag-quality-check[frame]`) |
181
+
182
+ `CaseResult` carries `.query`, `.metrics`, `.score`, `.excluded`, `.notes`,
183
+ `.failures`, `.ok`, `.summary()` and `.to_dict()`. A metric that could not be
184
+ computed is absent from `.metrics` and named in `.excluded`, never faked as
185
+ `0.0`.
186
+
187
+ ## CLI
188
+
189
+ ```bash
190
+ rag-quality-check cases.json # the summary
191
+ rag-quality-check cases.jsonl --k 3 --threshold 0.4
192
+ rag-quality-check cases.json --json > report.json # to_dict() as JSON
193
+ cat cases.json | rag-quality-check - --weakest 10
194
+ ```
195
+
196
+ `cases.json` is a list of the same case objects; `cases.jsonl` is one per line.
197
+ `--output PATH` writes the report (`.json` for the dict, otherwise the text).
198
+ `rag-quality-check --help` lists every metric and its one-line definition.
199
+
200
+ ## License
201
+
202
+ MIT
@@ -0,0 +1,175 @@
1
+ # rag-quality-check
2
+
3
+ Your RAG answers are only as good as what the retriever handed the model, and
4
+ "it feels better now" is not a number - this measures whether the right passages
5
+ actually came back, and whether the answer stayed inside them.
6
+
7
+ ## Install
8
+
9
+ ```bash
10
+ pip install rag-quality-check
11
+ ```
12
+
13
+ ## Quickstart
14
+
15
+ ```python
16
+ import rag_quality_check
17
+
18
+ cases = [
19
+ {"query": "reset my password", "retrieved": ["p_reset", "p_billing"], "relevant": ["p_reset"]},
20
+ {"query": "cancel my plan", "retrieved": ["p_faq", "p_billing"], "relevant": ["p_cancel"]},
21
+ ]
22
+ print(rag_quality_check.evaluate(cases, k=2).summary())
23
+ ```
24
+
25
+ ```
26
+ rag-quality-check: 2 case(s), top-2, threshold 0.5, similarity by lexical overlap.
27
+ retrieval:
28
+ precision_at_k 0.250 relevant items in the top k, divided by k (2 case(s))
29
+ recall_at_k 0.500 relevant items found, divided by all relevant ones (2 case(s))
30
+ mrr 0.500 1 / rank of the first relevant item, averaged over cases (2 case(s))
31
+ ndcg_at_k 0.500 DCG@k / ideal DCG@k, linear gains, log2 discount (2 case(s))
32
+ hit_rate 0.500 share of queries with a relevant item in the top k (2 case(s))
33
+ failures (1):
34
+ - case 1 'cancel my plan': nothing relevant in the top 2
35
+ weakest cases:
36
+ [1] 0.00 'cancel my plan'
37
+ [0] 0.90 'reset my password'
38
+ ```
39
+
40
+ ## What it measures
41
+
42
+ - **precision_at_k**, **recall_at_k**, **mrr**, **ndcg_at_k**, **hit_rate** - the
43
+ five retrieval numbers, each with its exact formula written out in its
44
+ docstring. Users compare these across systems, so an undocumented variant of
45
+ nDCG is worse than no nDCG at all.
46
+ - **groundedness** - the share of the answer's words that a retrieved passage
47
+ actually supports. This is the hallucination check. Sentences are weighted by
48
+ how many content words they hold, so a leading `"Yes."` cannot halve the score
49
+ of a paragraph copied straight out of the context.
50
+ - **citation_coverage** - the share of retrieved passages the answer drew on. A
51
+ low number beside a high groundedness means the retriever is returning padding.
52
+ - **answer_relevance** - how much of the query's own wording the answer repeats,
53
+ and **answer_correctness** when you pass a `ground_truth`.
54
+ - Every metric is in `0..1`. Every one of them is defined once, in one place.
55
+
56
+ It also tells you what it could *not* measure:
57
+
58
+ - No `relevant` ids on a case? Relevance is **estimated** from similarity to the
59
+ query, and the report says "ESTIMATED" every time it prints, on the report and
60
+ on the case. Estimated numbers are for spotting weak queries, not for
61
+ comparing two systems.
62
+ - A case whose `relevant` list is empty is a query you have labelled as having
63
+ no right answer. Every retrieval metric is **excluded** for it and the
64
+ exclusion is noted: retrieving nothing relevant is the wanted result there, so
65
+ scoring it `0.0` would drag the means down and fill `failures` with successes.
66
+ - Passed ids in `retrieved` instead of passage text? The retrieval metrics are
67
+ unaffected, but `groundedness` and `citation_coverage` are **excluded** with a
68
+ note rather than reported as `0.0` - an id carries no words for an answer to
69
+ be supported by, and scoring against one would report a hallucination that
70
+ never happened. Pass `{"id": ..., "text": ...}` to measure them.
71
+ - `answer_relevance` under the default word overlap is the share of the query's
72
+ content words the answer repeats, and a correct answer routinely repeats none
73
+ of them ("how much does the pro plan cost" answered by "It is 29 dollars per
74
+ seat per month." scores `0.00`). It is reported, noted when it is low, and
75
+ never counted as a failure unless you pass `embed=`, where it is a cosine
76
+ between meanings and does mean what it says.
77
+ - If no `relevant` id matches any retrieved id anywhere in the run, the report
78
+ says so, because ids compared in two different forms and a broken retriever
79
+ produce the same zeros.
80
+ - An empty `retrieved` list scores zeros, never a `ZeroDivisionError`.
81
+ - Asked for top-5 and only 3 came back? `precision_at_k` still divides by 5, and
82
+ the case says so in a note instead of silently shrinking `k`.
83
+ - Duplicate passages in `retrieved` are counted once, and the duplicate count is
84
+ noted.
85
+ - Unicode queries and passages work; CJK text is compared character by
86
+ character, since it is not written with spaces. One caveat, and only in
87
+ estimated mode: word overlap matches word forms, so an inflected language can
88
+ miss a passage that plainly answers the query (Hindi "बदलें" against "बदलने"
89
+ counts as a miss). Labelled `relevant` ids are unaffected, and `embed=` fixes
90
+ the estimate.
91
+ - 1000 cases score in about a second. No model downloads, no network, numpy only.
92
+
93
+ By default similarity is word overlap. Pass `embed=` any
94
+ `callable(list[str]) -> vectors` - a sentence-transformer `.encode`, an API
95
+ client, anything - and every similarity becomes cosine instead, so you get real
96
+ semantics without this package depending on a model:
97
+
98
+ ```python
99
+ report = rag_quality_check.evaluate(cases, embed=model.encode)
100
+ ```
101
+
102
+ Texts are de-duplicated and embedded in one batched call for the whole run.
103
+
104
+ ## API
105
+
106
+ ### `evaluate(cases, *, k=5, threshold=0.5, embed=None) -> RagReport`
107
+
108
+ `cases` is a list of dicts:
109
+
110
+ | key | required | meaning |
111
+ | --- | --- | --- |
112
+ | `query` | yes | the question that was asked |
113
+ | `retrieved` | yes | what came back, best first: passage strings, ids, or dicts with `id` and/or `text`. Ids alone are enough for the retrieval metrics; `groundedness` and `citation_coverage` need the text |
114
+ | `relevant` | no | ids that should have come back - a list, or `{id: gain}` for graded relevance. Without it, relevance is estimated; an empty list means the query has no right answer |
115
+ | `answer` | no | the generated answer; turns on the answer metrics |
116
+ | `ground_truth` | no | the reference answer; adds `answer_correctness` |
117
+
118
+ `k` is the cut-off and is always the denominator of `precision_at_k`.
119
+ `threshold` is the similarity at which two texts count as a match, and the score
120
+ below which a case is reported as a failure - with the single exception of
121
+ `answer_relevance` under word overlap, which is reported and never accused.
122
+
123
+ ### `evaluate_case(query, retrieved, **kw) -> CaseResult`
124
+
125
+ One query, same rules and the same numbers as a one-case `evaluate`. Takes
126
+ `relevant`, `answer`, `ground_truth`, `k`, `threshold` and `embed`.
127
+
128
+ ```python
129
+ case = rag_quality_check.evaluate_case(
130
+ "how do I reset my password",
131
+ ["To reset your password, open Settings and choose Reset."],
132
+ answer="Open Settings and choose Reset.",
133
+ k=1,
134
+ )
135
+ print(case.summary())
136
+ ```
137
+
138
+ ### `RagReport`
139
+
140
+ | member | what it is |
141
+ | --- | --- |
142
+ | `.metrics` | `dict[str, float]`, each metric averaged over the cases where it was defined |
143
+ | `.support` | how many cases went into each mean |
144
+ | `.per_case` | `list[CaseResult]` |
145
+ | `.weakest(n=5)` | the `n` worst cases, weakest first |
146
+ | `.failures` | `list[str]`, plain language: retrieved nothing, nothing relevant in the top k, answer not grounded, ... |
147
+ | `.notes` | everything that changed how a number was produced |
148
+ | `.estimated` / `.fully_estimated` | whether any / every case's relevance was guessed |
149
+ | `.similarity` | `"lexical overlap"` or `"embeddings"` |
150
+ | `.ok` | `True` when no case failed |
151
+ | `.summary()` | the human report above |
152
+ | `.to_dict()` | JSON-safe, per-case detail included |
153
+ | `.to_frame()` | one row per case as a `pandas.DataFrame` (needs `pip install rag-quality-check[frame]`) |
154
+
155
+ `CaseResult` carries `.query`, `.metrics`, `.score`, `.excluded`, `.notes`,
156
+ `.failures`, `.ok`, `.summary()` and `.to_dict()`. A metric that could not be
157
+ computed is absent from `.metrics` and named in `.excluded`, never faked as
158
+ `0.0`.
159
+
160
+ ## CLI
161
+
162
+ ```bash
163
+ rag-quality-check cases.json # the summary
164
+ rag-quality-check cases.jsonl --k 3 --threshold 0.4
165
+ rag-quality-check cases.json --json > report.json # to_dict() as JSON
166
+ cat cases.json | rag-quality-check - --weakest 10
167
+ ```
168
+
169
+ `cases.json` is a list of the same case objects; `cases.jsonl` is one per line.
170
+ `--output PATH` writes the report (`.json` for the dict, otherwise the text).
171
+ `rag-quality-check --help` lists every metric and its one-line definition.
172
+
173
+ ## License
174
+
175
+ MIT
@@ -0,0 +1,50 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "rag-quality-check"
7
+ version = "0.1.0"
8
+ description = "Measure whether a retrieval system is actually retrieving the right things"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Pranay Mahendrakar" }]
14
+ keywords = [
15
+ "rag",
16
+ "retrieval",
17
+ "evaluation",
18
+ "ndcg",
19
+ "mrr",
20
+ "groundedness",
21
+ "information-retrieval",
22
+ "llm",
23
+ ]
24
+ classifiers = [
25
+ "Development Status :: 4 - Beta",
26
+ "Intended Audience :: Developers",
27
+ "Intended Audience :: Science/Research",
28
+ "Programming Language :: Python :: 3",
29
+ "Programming Language :: Python :: 3 :: Only",
30
+ "Operating System :: OS Independent",
31
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
32
+ "Topic :: Text Processing :: Linguistic",
33
+ ]
34
+ dependencies = [
35
+ "numpy>=1.23",
36
+ ]
37
+
38
+ [project.optional-dependencies]
39
+ frame = ["pandas>=1.3"]
40
+ dev = ["pytest>=7", "pandas>=1.3"]
41
+
42
+ [project.scripts]
43
+ rag-quality-check = "rag_quality_check.cli:main"
44
+
45
+ [project.urls]
46
+ Homepage = "https://pypi.org/project/rag-quality-check/"
47
+ Author = "https://pypi.org/user/pranaymahendrakar/"
48
+
49
+ [tool.hatch.build.targets.wheel]
50
+ packages = ["src/rag_quality_check"]
@@ -0,0 +1,60 @@
1
+ """rag-quality-check: measure whether your retriever is retrieving the right things.
2
+
3
+ Quick use::
4
+
5
+ import rag_quality_check
6
+
7
+ report = rag_quality_check.evaluate([
8
+ {"query": "how do I reset my password",
9
+ "retrieved": ["p_reset", "p_billing"],
10
+ "relevant": ["p_reset"]},
11
+ ])
12
+ print(report.summary())
13
+
14
+ Every metric is in ``0..1`` and its exact formula is in the docstring of the
15
+ function that produces it, because an undocumented variant of nDCG is worse than
16
+ no nDCG at all. When a case has no ``relevant`` ids the numbers are *estimated*
17
+ from similarity rather than measured, and the report says so every time.
18
+ """
19
+ from ._core import (
20
+ ANSWER_METRICS,
21
+ METRIC_HELP,
22
+ RETRIEVAL_METRICS,
23
+ CaseResult,
24
+ RagReport,
25
+ evaluate,
26
+ evaluate_case,
27
+ )
28
+ from ._metrics import (
29
+ dcg,
30
+ hit_rate,
31
+ mrr,
32
+ ndcg_at_k,
33
+ precision_at_k,
34
+ recall_at_k,
35
+ )
36
+ from ._text import Similarity, cosine, coverage, split_sentences, tokenize
37
+
38
+ __version__ = "0.1.0"
39
+
40
+ __all__ = [
41
+ "ANSWER_METRICS",
42
+ "METRIC_HELP",
43
+ "RETRIEVAL_METRICS",
44
+ "CaseResult",
45
+ "RagReport",
46
+ "Similarity",
47
+ "cosine",
48
+ "coverage",
49
+ "dcg",
50
+ "evaluate",
51
+ "evaluate_case",
52
+ "hit_rate",
53
+ "mrr",
54
+ "ndcg_at_k",
55
+ "precision_at_k",
56
+ "recall_at_k",
57
+ "split_sentences",
58
+ "tokenize",
59
+ "__version__",
60
+ ]