rag-quality-check 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_quality_check-0.1.0/.gitignore +30 -0
- rag_quality_check-0.1.0/LICENSE +21 -0
- rag_quality_check-0.1.0/PKG-INFO +202 -0
- rag_quality_check-0.1.0/README.md +175 -0
- rag_quality_check-0.1.0/pyproject.toml +50 -0
- rag_quality_check-0.1.0/src/rag_quality_check/__init__.py +60 -0
- rag_quality_check-0.1.0/src/rag_quality_check/_core.py +1044 -0
- rag_quality_check-0.1.0/src/rag_quality_check/_metrics.py +116 -0
- rag_quality_check-0.1.0/src/rag_quality_check/_text.py +332 -0
- rag_quality_check-0.1.0/src/rag_quality_check/cli.py +205 -0
- rag_quality_check-0.1.0/tests/test_cli.py +187 -0
- rag_quality_check-0.1.0/tests/test_core.py +49 -0
- rag_quality_check-0.1.0/tests/test_evaluate.py +749 -0
- rag_quality_check-0.1.0/tests/test_metrics.py +77 -0
- rag_quality_check-0.1.0/tests/test_text.py +159 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Virtual environments and scratch space used while building. Never committed:
|
|
2
|
+
# they are large, machine-specific, and rebuilt from pyproject.toml anyway.
|
|
3
|
+
_venvs/
|
|
4
|
+
_proof/
|
|
5
|
+
.venv*/
|
|
6
|
+
|
|
7
|
+
# Build output. Wheels are built by the release workflow from this source, so a
|
|
8
|
+
# wheel in git could silently differ from the code beside it.
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
src/*.egg-info/
|
|
13
|
+
|
|
14
|
+
# Python noise
|
|
15
|
+
__pycache__/
|
|
16
|
+
*.py[cod]
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.mypy_cache/
|
|
19
|
+
.ruff_cache/
|
|
20
|
+
|
|
21
|
+
# Local verification state, not source
|
|
22
|
+
verify_results.json
|
|
23
|
+
publish-log.txt
|
|
24
|
+
published.json
|
|
25
|
+
|
|
26
|
+
# Credentials. None of these belong here, and this line is the backstop.
|
|
27
|
+
.pypirc
|
|
28
|
+
.env
|
|
29
|
+
*.pem
|
|
30
|
+
*.key
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pranay Mahendrakar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: rag-quality-check
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Measure whether a retrieval system is actually retrieving the right things
|
|
5
|
+
Project-URL: Homepage, https://pypi.org/project/rag-quality-check/
|
|
6
|
+
Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
|
|
7
|
+
Author: Pranay Mahendrakar
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: evaluation,groundedness,information-retrieval,llm,mrr,ndcg,rag,retrieval
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Requires-Dist: numpy>=1.23
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pandas>=1.3; extra == 'dev'
|
|
23
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
24
|
+
Provides-Extra: frame
|
|
25
|
+
Requires-Dist: pandas>=1.3; extra == 'frame'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# rag-quality-check
|
|
29
|
+
|
|
30
|
+
Your RAG answers are only as good as what the retriever handed the model, and
|
|
31
|
+
"it feels better now" is not a number - this measures whether the right passages
|
|
32
|
+
actually came back, and whether the answer stayed inside them.
|
|
33
|
+
|
|
34
|
+
## Install
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install rag-quality-check
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Quickstart
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
import rag_quality_check
|
|
44
|
+
|
|
45
|
+
cases = [
|
|
46
|
+
{"query": "reset my password", "retrieved": ["p_reset", "p_billing"], "relevant": ["p_reset"]},
|
|
47
|
+
{"query": "cancel my plan", "retrieved": ["p_faq", "p_billing"], "relevant": ["p_cancel"]},
|
|
48
|
+
]
|
|
49
|
+
print(rag_quality_check.evaluate(cases, k=2).summary())
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
rag-quality-check: 2 case(s), top-2, threshold 0.5, similarity by lexical overlap.
|
|
54
|
+
retrieval:
|
|
55
|
+
precision_at_k 0.250 relevant items in the top k, divided by k (2 case(s))
|
|
56
|
+
recall_at_k 0.500 relevant items found, divided by all relevant ones (2 case(s))
|
|
57
|
+
mrr 0.500 1 / rank of the first relevant item, averaged over cases (2 case(s))
|
|
58
|
+
ndcg_at_k 0.500 DCG@k / ideal DCG@k, linear gains, log2 discount (2 case(s))
|
|
59
|
+
hit_rate 0.500 share of queries with a relevant item in the top k (2 case(s))
|
|
60
|
+
failures (1):
|
|
61
|
+
- case 1 'cancel my plan': nothing relevant in the top 2
|
|
62
|
+
weakest cases:
|
|
63
|
+
[1] 0.00 'cancel my plan'
|
|
64
|
+
[0] 0.90 'reset my password'
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## What it measures
|
|
68
|
+
|
|
69
|
+
- **precision_at_k**, **recall_at_k**, **mrr**, **ndcg_at_k**, **hit_rate** - the
|
|
70
|
+
five retrieval numbers, each with its exact formula written out in its
|
|
71
|
+
docstring. Users compare these across systems, so an undocumented variant of
|
|
72
|
+
nDCG is worse than no nDCG at all.
|
|
73
|
+
- **groundedness** - the share of the answer's words that a retrieved passage
|
|
74
|
+
actually supports. This is the hallucination check. Sentences are weighted by
|
|
75
|
+
how many content words they hold, so a leading `"Yes."` cannot halve the score
|
|
76
|
+
of a paragraph copied straight out of the context.
|
|
77
|
+
- **citation_coverage** - the share of retrieved passages the answer drew on. A
|
|
78
|
+
low number beside a high groundedness means the retriever is returning padding.
|
|
79
|
+
- **answer_relevance** - how much of the query's own wording the answer repeats,
|
|
80
|
+
and **answer_correctness** when you pass a `ground_truth`.
|
|
81
|
+
- Every metric is in `0..1`. Every one of them is defined once, in one place.
|
|
82
|
+
|
|
83
|
+
It also tells you what it could *not* measure:
|
|
84
|
+
|
|
85
|
+
- No `relevant` ids on a case? Relevance is **estimated** from similarity to the
|
|
86
|
+
query, and the report says "ESTIMATED" every time it prints, on the report and
|
|
87
|
+
on the case. Estimated numbers are for spotting weak queries, not for
|
|
88
|
+
comparing two systems.
|
|
89
|
+
- A case whose `relevant` list is empty is a query you have labelled as having
|
|
90
|
+
no right answer. Every retrieval metric is **excluded** for it and the
|
|
91
|
+
exclusion is noted: retrieving nothing relevant is the wanted result there, so
|
|
92
|
+
scoring it `0.0` would drag the means down and fill `failures` with successes.
|
|
93
|
+
- Passed ids in `retrieved` instead of passage text? The retrieval metrics are
|
|
94
|
+
unaffected, but `groundedness` and `citation_coverage` are **excluded** with a
|
|
95
|
+
note rather than reported as `0.0` - an id carries no words for an answer to
|
|
96
|
+
be supported by, and scoring against one would report a hallucination that
|
|
97
|
+
never happened. Pass `{"id": ..., "text": ...}` to measure them.
|
|
98
|
+
- `answer_relevance` under the default word overlap is the share of the query's
|
|
99
|
+
content words the answer repeats, and a correct answer routinely repeats none
|
|
100
|
+
of them ("how much does the pro plan cost" answered by "It is 29 dollars per
|
|
101
|
+
seat per month." scores `0.00`). It is reported, noted when it is low, and
|
|
102
|
+
never counted as a failure unless you pass `embed=`, where it is a cosine
|
|
103
|
+
between meanings and does mean what it says.
|
|
104
|
+
- If no `relevant` id matches any retrieved id anywhere in the run, the report
|
|
105
|
+
says so, because ids compared in two different forms and a broken retriever
|
|
106
|
+
produce the same zeros.
|
|
107
|
+
- An empty `retrieved` list scores zeros, never a `ZeroDivisionError`.
|
|
108
|
+
- Asked for top-5 and only 3 came back? `precision_at_k` still divides by 5, and
|
|
109
|
+
the case says so in a note instead of silently shrinking `k`.
|
|
110
|
+
- Duplicate passages in `retrieved` are counted once, and the duplicate count is
|
|
111
|
+
noted.
|
|
112
|
+
- Unicode queries and passages work; CJK text is compared character by
|
|
113
|
+
character, since it is not written with spaces. One caveat, and only in
|
|
114
|
+
estimated mode: word overlap matches word forms, so an inflected language can
|
|
115
|
+
miss a passage that plainly answers the query (Hindi "बदलें" against "बदलने"
|
|
116
|
+
counts as a miss). Labelled `relevant` ids are unaffected, and `embed=` fixes
|
|
117
|
+
the estimate.
|
|
118
|
+
- 1000 cases score in about a second. No model downloads, no network, numpy only.
|
|
119
|
+
|
|
120
|
+
By default similarity is word overlap. Pass `embed=` any
|
|
121
|
+
`callable(list[str]) -> vectors` - a sentence-transformer `.encode`, an API
|
|
122
|
+
client, anything - and every similarity becomes cosine instead, so you get real
|
|
123
|
+
semantics without this package depending on a model:
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
report = rag_quality_check.evaluate(cases, embed=model.encode)
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Texts are de-duplicated and embedded in one batched call for the whole run.
|
|
130
|
+
|
|
131
|
+
## API
|
|
132
|
+
|
|
133
|
+
### `evaluate(cases, *, k=5, threshold=0.5, embed=None) -> RagReport`
|
|
134
|
+
|
|
135
|
+
`cases` is a list of dicts:
|
|
136
|
+
|
|
137
|
+
| key | required | meaning |
|
|
138
|
+
| --- | --- | --- |
|
|
139
|
+
| `query` | yes | the question that was asked |
|
|
140
|
+
| `retrieved` | yes | what came back, best first: passage strings, ids, or dicts with `id` and/or `text`. Ids alone are enough for the retrieval metrics; `groundedness` and `citation_coverage` need the text |
|
|
141
|
+
| `relevant` | no | ids that should have come back - a list, or `{id: gain}` for graded relevance. Without it, relevance is estimated; an empty list means the query has no right answer |
|
|
142
|
+
| `answer` | no | the generated answer; turns on the answer metrics |
|
|
143
|
+
| `ground_truth` | no | the reference answer; adds `answer_correctness` |
|
|
144
|
+
|
|
145
|
+
`k` is the cut-off and is always the denominator of `precision_at_k`.
|
|
146
|
+
`threshold` is the similarity at which two texts count as a match, and the score
|
|
147
|
+
below which a case is reported as a failure - with the single exception of
|
|
148
|
+
`answer_relevance` under word overlap, which is reported and never accused.
|
|
149
|
+
|
|
150
|
+
### `evaluate_case(query, retrieved, **kw) -> CaseResult`
|
|
151
|
+
|
|
152
|
+
One query, same rules and the same numbers as a one-case `evaluate`. Takes
|
|
153
|
+
`relevant`, `answer`, `ground_truth`, `k`, `threshold` and `embed`.
|
|
154
|
+
|
|
155
|
+
```python
|
|
156
|
+
case = rag_quality_check.evaluate_case(
|
|
157
|
+
"how do I reset my password",
|
|
158
|
+
["To reset your password, open Settings and choose Reset."],
|
|
159
|
+
answer="Open Settings and choose Reset.",
|
|
160
|
+
k=1,
|
|
161
|
+
)
|
|
162
|
+
print(case.summary())
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
### `RagReport`
|
|
166
|
+
|
|
167
|
+
| member | what it is |
|
|
168
|
+
| --- | --- |
|
|
169
|
+
| `.metrics` | `dict[str, float]`, each metric averaged over the cases where it was defined |
|
|
170
|
+
| `.support` | how many cases went into each mean |
|
|
171
|
+
| `.per_case` | `list[CaseResult]` |
|
|
172
|
+
| `.weakest(n=5)` | the `n` worst cases, weakest first |
|
|
173
|
+
| `.failures` | `list[str]`, plain language: retrieved nothing, nothing relevant in the top k, answer not grounded, ... |
|
|
174
|
+
| `.notes` | everything that changed how a number was produced |
|
|
175
|
+
| `.estimated` / `.fully_estimated` | whether any / every case's relevance was guessed |
|
|
176
|
+
| `.similarity` | `"lexical overlap"` or `"embeddings"` |
|
|
177
|
+
| `.ok` | `True` when no case failed |
|
|
178
|
+
| `.summary()` | the human report above |
|
|
179
|
+
| `.to_dict()` | JSON-safe, per-case detail included |
|
|
180
|
+
| `.to_frame()` | one row per case as a `pandas.DataFrame` (needs `pip install rag-quality-check[frame]`) |
|
|
181
|
+
|
|
182
|
+
`CaseResult` carries `.query`, `.metrics`, `.score`, `.excluded`, `.notes`,
|
|
183
|
+
`.failures`, `.ok`, `.summary()` and `.to_dict()`. A metric that could not be
|
|
184
|
+
computed is absent from `.metrics` and named in `.excluded`, never faked as
|
|
185
|
+
`0.0`.
|
|
186
|
+
|
|
187
|
+
## CLI
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
rag-quality-check cases.json # the summary
|
|
191
|
+
rag-quality-check cases.jsonl --k 3 --threshold 0.4
|
|
192
|
+
rag-quality-check cases.json --json > report.json # to_dict() as JSON
|
|
193
|
+
cat cases.json | rag-quality-check - --weakest 10
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
`cases.json` is a list of the same case objects; `cases.jsonl` is one per line.
|
|
197
|
+
`--output PATH` writes the report (`.json` for the dict, otherwise the text).
|
|
198
|
+
`rag-quality-check --help` lists every metric and its one-line definition.
|
|
199
|
+
|
|
200
|
+
## License
|
|
201
|
+
|
|
202
|
+
MIT
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
# rag-quality-check
|
|
2
|
+
|
|
3
|
+
Your RAG answers are only as good as what the retriever handed the model, and
|
|
4
|
+
"it feels better now" is not a number - this measures whether the right passages
|
|
5
|
+
actually came back, and whether the answer stayed inside them.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install rag-quality-check
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Quickstart
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import rag_quality_check
|
|
17
|
+
|
|
18
|
+
cases = [
|
|
19
|
+
{"query": "reset my password", "retrieved": ["p_reset", "p_billing"], "relevant": ["p_reset"]},
|
|
20
|
+
{"query": "cancel my plan", "retrieved": ["p_faq", "p_billing"], "relevant": ["p_cancel"]},
|
|
21
|
+
]
|
|
22
|
+
print(rag_quality_check.evaluate(cases, k=2).summary())
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
```
|
|
26
|
+
rag-quality-check: 2 case(s), top-2, threshold 0.5, similarity by lexical overlap.
|
|
27
|
+
retrieval:
|
|
28
|
+
precision_at_k 0.250 relevant items in the top k, divided by k (2 case(s))
|
|
29
|
+
recall_at_k 0.500 relevant items found, divided by all relevant ones (2 case(s))
|
|
30
|
+
mrr 0.500 1 / rank of the first relevant item, averaged over cases (2 case(s))
|
|
31
|
+
ndcg_at_k 0.500 DCG@k / ideal DCG@k, linear gains, log2 discount (2 case(s))
|
|
32
|
+
hit_rate 0.500 share of queries with a relevant item in the top k (2 case(s))
|
|
33
|
+
failures (1):
|
|
34
|
+
- case 1 'cancel my plan': nothing relevant in the top 2
|
|
35
|
+
weakest cases:
|
|
36
|
+
[1] 0.00 'cancel my plan'
|
|
37
|
+
[0] 0.90 'reset my password'
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## What it measures
|
|
41
|
+
|
|
42
|
+
- **precision_at_k**, **recall_at_k**, **mrr**, **ndcg_at_k**, **hit_rate** - the
|
|
43
|
+
five retrieval numbers, each with its exact formula written out in its
|
|
44
|
+
docstring. Users compare these across systems, so an undocumented variant of
|
|
45
|
+
nDCG is worse than no nDCG at all.
|
|
46
|
+
- **groundedness** - the share of the answer's words that a retrieved passage
|
|
47
|
+
actually supports. This is the hallucination check. Sentences are weighted by
|
|
48
|
+
how many content words they hold, so a leading `"Yes."` cannot halve the score
|
|
49
|
+
of a paragraph copied straight out of the context.
|
|
50
|
+
- **citation_coverage** - the share of retrieved passages the answer drew on. A
|
|
51
|
+
low number beside a high groundedness means the retriever is returning padding.
|
|
52
|
+
- **answer_relevance** - how much of the query's own wording the answer repeats,
|
|
53
|
+
and **answer_correctness** when you pass a `ground_truth`.
|
|
54
|
+
- Every metric is in `0..1`. Every one of them is defined once, in one place.
|
|
55
|
+
|
|
56
|
+
It also tells you what it could *not* measure:
|
|
57
|
+
|
|
58
|
+
- No `relevant` ids on a case? Relevance is **estimated** from similarity to the
|
|
59
|
+
query, and the report says "ESTIMATED" every time it prints, on the report and
|
|
60
|
+
on the case. Estimated numbers are for spotting weak queries, not for
|
|
61
|
+
comparing two systems.
|
|
62
|
+
- A case whose `relevant` list is empty is a query you have labelled as having
|
|
63
|
+
no right answer. Every retrieval metric is **excluded** for it and the
|
|
64
|
+
exclusion is noted: retrieving nothing relevant is the wanted result there, so
|
|
65
|
+
scoring it `0.0` would drag the means down and fill `failures` with successes.
|
|
66
|
+
- Passed ids in `retrieved` instead of passage text? The retrieval metrics are
|
|
67
|
+
unaffected, but `groundedness` and `citation_coverage` are **excluded** with a
|
|
68
|
+
note rather than reported as `0.0` - an id carries no words for an answer to
|
|
69
|
+
be supported by, and scoring against one would report a hallucination that
|
|
70
|
+
never happened. Pass `{"id": ..., "text": ...}` to measure them.
|
|
71
|
+
- `answer_relevance` under the default word overlap is the share of the query's
|
|
72
|
+
content words the answer repeats, and a correct answer routinely repeats none
|
|
73
|
+
of them ("how much does the pro plan cost" answered by "It is 29 dollars per
|
|
74
|
+
seat per month." scores `0.00`). It is reported, noted when it is low, and
|
|
75
|
+
never counted as a failure unless you pass `embed=`, where it is a cosine
|
|
76
|
+
between meanings and does mean what it says.
|
|
77
|
+
- If no `relevant` id matches any retrieved id anywhere in the run, the report
|
|
78
|
+
says so, because ids compared in two different forms and a broken retriever
|
|
79
|
+
produce the same zeros.
|
|
80
|
+
- An empty `retrieved` list scores zeros, never a `ZeroDivisionError`.
|
|
81
|
+
- Asked for top-5 and only 3 came back? `precision_at_k` still divides by 5, and
|
|
82
|
+
the case says so in a note instead of silently shrinking `k`.
|
|
83
|
+
- Duplicate passages in `retrieved` are counted once, and the duplicate count is
|
|
84
|
+
noted.
|
|
85
|
+
- Unicode queries and passages work; CJK text is compared character by
|
|
86
|
+
character, since it is not written with spaces. One caveat, and only in
|
|
87
|
+
estimated mode: word overlap matches word forms, so an inflected language can
|
|
88
|
+
miss a passage that plainly answers the query (Hindi "बदलें" against "बदलने"
|
|
89
|
+
counts as a miss). Labelled `relevant` ids are unaffected, and `embed=` fixes
|
|
90
|
+
the estimate.
|
|
91
|
+
- 1000 cases score in about a second. No model downloads, no network, numpy only.
|
|
92
|
+
|
|
93
|
+
By default similarity is word overlap. Pass `embed=` any
|
|
94
|
+
`callable(list[str]) -> vectors` - a sentence-transformer `.encode`, an API
|
|
95
|
+
client, anything - and every similarity becomes cosine instead, so you get real
|
|
96
|
+
semantics without this package depending on a model:
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
report = rag_quality_check.evaluate(cases, embed=model.encode)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Texts are de-duplicated and embedded in one batched call for the whole run.
|
|
103
|
+
|
|
104
|
+
## API
|
|
105
|
+
|
|
106
|
+
### `evaluate(cases, *, k=5, threshold=0.5, embed=None) -> RagReport`
|
|
107
|
+
|
|
108
|
+
`cases` is a list of dicts:
|
|
109
|
+
|
|
110
|
+
| key | required | meaning |
|
|
111
|
+
| --- | --- | --- |
|
|
112
|
+
| `query` | yes | the question that was asked |
|
|
113
|
+
| `retrieved` | yes | what came back, best first: passage strings, ids, or dicts with `id` and/or `text`. Ids alone are enough for the retrieval metrics; `groundedness` and `citation_coverage` need the text |
|
|
114
|
+
| `relevant` | no | ids that should have come back - a list, or `{id: gain}` for graded relevance. Without it, relevance is estimated; an empty list means the query has no right answer |
|
|
115
|
+
| `answer` | no | the generated answer; turns on the answer metrics |
|
|
116
|
+
| `ground_truth` | no | the reference answer; adds `answer_correctness` |
|
|
117
|
+
|
|
118
|
+
`k` is the cut-off and is always the denominator of `precision_at_k`.
|
|
119
|
+
`threshold` is the similarity at which two texts count as a match, and the score
|
|
120
|
+
below which a case is reported as a failure - with the single exception of
|
|
121
|
+
`answer_relevance` under word overlap, which is reported and never accused.
|
|
122
|
+
|
|
123
|
+
### `evaluate_case(query, retrieved, **kw) -> CaseResult`
|
|
124
|
+
|
|
125
|
+
One query, same rules and the same numbers as a one-case `evaluate`. Takes
|
|
126
|
+
`relevant`, `answer`, `ground_truth`, `k`, `threshold` and `embed`.
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
case = rag_quality_check.evaluate_case(
|
|
130
|
+
"how do I reset my password",
|
|
131
|
+
["To reset your password, open Settings and choose Reset."],
|
|
132
|
+
answer="Open Settings and choose Reset.",
|
|
133
|
+
k=1,
|
|
134
|
+
)
|
|
135
|
+
print(case.summary())
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
### `RagReport`
|
|
139
|
+
|
|
140
|
+
| member | what it is |
|
|
141
|
+
| --- | --- |
|
|
142
|
+
| `.metrics` | `dict[str, float]`, each metric averaged over the cases where it was defined |
|
|
143
|
+
| `.support` | how many cases went into each mean |
|
|
144
|
+
| `.per_case` | `list[CaseResult]` |
|
|
145
|
+
| `.weakest(n=5)` | the `n` worst cases, weakest first |
|
|
146
|
+
| `.failures` | `list[str]`, plain language: retrieved nothing, nothing relevant in the top k, answer not grounded, ... |
|
|
147
|
+
| `.notes` | everything that changed how a number was produced |
|
|
148
|
+
| `.estimated` / `.fully_estimated` | whether any / every case's relevance was guessed |
|
|
149
|
+
| `.similarity` | `"lexical overlap"` or `"embeddings"` |
|
|
150
|
+
| `.ok` | `True` when no case failed |
|
|
151
|
+
| `.summary()` | the human report above |
|
|
152
|
+
| `.to_dict()` | JSON-safe, per-case detail included |
|
|
153
|
+
| `.to_frame()` | one row per case as a `pandas.DataFrame` (needs `pip install rag-quality-check[frame]`) |
|
|
154
|
+
|
|
155
|
+
`CaseResult` carries `.query`, `.metrics`, `.score`, `.excluded`, `.notes`,
|
|
156
|
+
`.failures`, `.ok`, `.summary()` and `.to_dict()`. A metric that could not be
|
|
157
|
+
computed is absent from `.metrics` and named in `.excluded`, never faked as
|
|
158
|
+
`0.0`.
|
|
159
|
+
|
|
160
|
+
## CLI
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
rag-quality-check cases.json # the summary
|
|
164
|
+
rag-quality-check cases.jsonl --k 3 --threshold 0.4
|
|
165
|
+
rag-quality-check cases.json --json > report.json # to_dict() as JSON
|
|
166
|
+
cat cases.json | rag-quality-check - --weakest 10
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
`cases.json` is a list of the same case objects; `cases.jsonl` is one per line.
|
|
170
|
+
`--output PATH` writes the report (`.json` for the dict, otherwise the text).
|
|
171
|
+
`rag-quality-check --help` lists every metric and its one-line definition.
|
|
172
|
+
|
|
173
|
+
## License
|
|
174
|
+
|
|
175
|
+
MIT
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "rag-quality-check"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Measure whether a retrieval system is actually retrieving the right things"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Pranay Mahendrakar" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"rag",
|
|
16
|
+
"retrieval",
|
|
17
|
+
"evaluation",
|
|
18
|
+
"ndcg",
|
|
19
|
+
"mrr",
|
|
20
|
+
"groundedness",
|
|
21
|
+
"information-retrieval",
|
|
22
|
+
"llm",
|
|
23
|
+
]
|
|
24
|
+
classifiers = [
|
|
25
|
+
"Development Status :: 4 - Beta",
|
|
26
|
+
"Intended Audience :: Developers",
|
|
27
|
+
"Intended Audience :: Science/Research",
|
|
28
|
+
"Programming Language :: Python :: 3",
|
|
29
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
30
|
+
"Operating System :: OS Independent",
|
|
31
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
32
|
+
"Topic :: Text Processing :: Linguistic",
|
|
33
|
+
]
|
|
34
|
+
dependencies = [
|
|
35
|
+
"numpy>=1.23",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
frame = ["pandas>=1.3"]
|
|
40
|
+
dev = ["pytest>=7", "pandas>=1.3"]
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
rag-quality-check = "rag_quality_check.cli:main"
|
|
44
|
+
|
|
45
|
+
[project.urls]
|
|
46
|
+
Homepage = "https://pypi.org/project/rag-quality-check/"
|
|
47
|
+
Author = "https://pypi.org/user/pranaymahendrakar/"
|
|
48
|
+
|
|
49
|
+
[tool.hatch.build.targets.wheel]
|
|
50
|
+
packages = ["src/rag_quality_check"]
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""rag-quality-check: measure whether your retriever is retrieving the right things.
|
|
2
|
+
|
|
3
|
+
Quick use::
|
|
4
|
+
|
|
5
|
+
import rag_quality_check
|
|
6
|
+
|
|
7
|
+
report = rag_quality_check.evaluate([
|
|
8
|
+
{"query": "how do I reset my password",
|
|
9
|
+
"retrieved": ["p_reset", "p_billing"],
|
|
10
|
+
"relevant": ["p_reset"]},
|
|
11
|
+
])
|
|
12
|
+
print(report.summary())
|
|
13
|
+
|
|
14
|
+
Every metric is in ``0..1`` and its exact formula is in the docstring of the
|
|
15
|
+
function that produces it, because an undocumented variant of nDCG is worse than
|
|
16
|
+
no nDCG at all. When a case has no ``relevant`` ids the numbers are *estimated*
|
|
17
|
+
from similarity rather than measured, and the report says so every time.
|
|
18
|
+
"""
|
|
19
|
+
from ._core import (
|
|
20
|
+
ANSWER_METRICS,
|
|
21
|
+
METRIC_HELP,
|
|
22
|
+
RETRIEVAL_METRICS,
|
|
23
|
+
CaseResult,
|
|
24
|
+
RagReport,
|
|
25
|
+
evaluate,
|
|
26
|
+
evaluate_case,
|
|
27
|
+
)
|
|
28
|
+
from ._metrics import (
|
|
29
|
+
dcg,
|
|
30
|
+
hit_rate,
|
|
31
|
+
mrr,
|
|
32
|
+
ndcg_at_k,
|
|
33
|
+
precision_at_k,
|
|
34
|
+
recall_at_k,
|
|
35
|
+
)
|
|
36
|
+
from ._text import Similarity, cosine, coverage, split_sentences, tokenize
|
|
37
|
+
|
|
38
|
+
__version__ = "0.1.0"
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"ANSWER_METRICS",
|
|
42
|
+
"METRIC_HELP",
|
|
43
|
+
"RETRIEVAL_METRICS",
|
|
44
|
+
"CaseResult",
|
|
45
|
+
"RagReport",
|
|
46
|
+
"Similarity",
|
|
47
|
+
"cosine",
|
|
48
|
+
"coverage",
|
|
49
|
+
"dcg",
|
|
50
|
+
"evaluate",
|
|
51
|
+
"evaluate_case",
|
|
52
|
+
"hit_rate",
|
|
53
|
+
"mrr",
|
|
54
|
+
"ndcg_at_k",
|
|
55
|
+
"precision_at_k",
|
|
56
|
+
"recall_at_k",
|
|
57
|
+
"split_sentences",
|
|
58
|
+
"tokenize",
|
|
59
|
+
"__version__",
|
|
60
|
+
]
|