microrag 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {microrag-0.2.2 → microrag-0.2.4}/LICENSE +1 -1
- {microrag-0.2.2 → microrag-0.2.4}/PKG-INFO +32 -4
- {microrag-0.2.2 → microrag-0.2.4}/README.md +30 -2
- {microrag-0.2.2 → microrag-0.2.4}/pyproject.toml +1 -1
- {microrag-0.2.2 → microrag-0.2.4}/.github/workflows/ci.yml +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/.github/workflows/publish.yml +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/.gitignore +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/Makefile +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/examples/advanced_config.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/examples/basic_usage.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/examples/faq_search.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/__init__.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/config.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/core.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/__init__.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/base.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/factory.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/fastembed.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/sentence_transformers.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/exceptions.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/models.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/query_processor.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/search/__init__.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/search/bm25.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/search/hybrid.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/stopwords.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/storage/__init__.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/storage/base.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/storage/duckdb.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/src/microrag/utils.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/__init__.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/conftest.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/test_bm25.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/test_config.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/test_integration.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/test_models.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/test_query_processor.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/test_storage.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/tests/test_utils.py +0 -0
- {microrag-0.2.2 → microrag-0.2.4}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: microrag
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: A feature-rich, universal RAG library for Python with ONNX-backed embeddings and DuckDB storage
|
|
5
5
|
Author-email: Pavel Liashkov <pavel.liashkov@protonamil.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -188,6 +188,34 @@ Each search method has different strengths:
|
|
|
188
188
|
|
|
189
189
|
By combining all three with RRF fusion, MicroRAG achieves better recall and precision than any single method alone.
|
|
190
190
|
|
|
191
|
+
### Filtering Irrelevant Results
|
|
192
|
+
|
|
193
|
+
By default, search returns the top-k results regardless of relevance. For queries like "111111" or random gibberish, the system will still return documents (just with lower scores). To filter out irrelevant results, use `similarity_threshold`.
|
|
194
|
+
|
|
195
|
+
**Understanding RRF scores:** RRF fusion produces scores typically in the 0.01-0.03 range, not 0-1 like raw cosine similarity. This is because RRF scores are based on rank positions: `score = Σ weight / (k + rank)`.
|
|
196
|
+
|
|
197
|
+
**Finding the right threshold:**
|
|
198
|
+
|
|
199
|
+
```python
|
|
200
|
+
# Test with your data to find appropriate threshold
|
|
201
|
+
results = rag.search("relevant query", threshold=0.0)
|
|
202
|
+
print(f"Relevant score: {results[0].score}") # e.g., 0.016
|
|
203
|
+
|
|
204
|
+
results = rag.search("gibberish123", threshold=0.0)
|
|
205
|
+
print(f"Irrelevant score: {results[0].score}") # e.g., 0.011
|
|
206
|
+
|
|
207
|
+
# Set threshold between irrelevant and relevant scores
|
|
208
|
+
config = RAGConfig(
|
|
209
|
+
model_path="...",
|
|
210
|
+
similarity_threshold=0.014, # Filters gibberish, keeps relevant
|
|
211
|
+
)
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
**Typical thresholds:**
|
|
215
|
+
- `0.0` - Return all results (no filtering)
|
|
216
|
+
- `0.010-0.015` - Filter obvious gibberish while keeping most relevant results
|
|
217
|
+
- `0.015-0.020` - Stricter filtering, may reduce recall for edge cases
|
|
218
|
+
|
|
191
219
|
## Configuration
|
|
192
220
|
|
|
193
221
|
```python
|
|
@@ -209,7 +237,7 @@ config = RAGConfig(
|
|
|
209
237
|
# Search
|
|
210
238
|
hybrid_enabled=True, # Enable hybrid search
|
|
211
239
|
hybrid_alpha=0.7, # Semantic weight (0-1)
|
|
212
|
-
similarity_threshold=0.
|
|
240
|
+
similarity_threshold=0.014, # Min score threshold (RRF scores are ~0.01-0.03)
|
|
213
241
|
|
|
214
242
|
# Query processing
|
|
215
243
|
abbreviations={"ML": "machine learning"}, # Query expansion
|
|
@@ -241,7 +269,7 @@ config = RAGConfig(
|
|
|
241
269
|
**Search:**
|
|
242
270
|
- `hybrid_enabled` (bool, default: True) - Enable hybrid search
|
|
243
271
|
- `hybrid_alpha` (float, default: 0.7) - Semantic weight in fusion (0-1)
|
|
244
|
-
- `similarity_threshold` (float, default: 0.4) - Minimum score to return
|
|
272
|
+
- `similarity_threshold` (float, default: 0.4) - Minimum score to return (see [Filtering Irrelevant Results](#filtering-irrelevant-results))
|
|
245
273
|
|
|
246
274
|
**Query Processing:**
|
|
247
275
|
- `abbreviations` (dict, default: None) - Query expansion mapping
|
|
@@ -152,6 +152,34 @@ Each search method has different strengths:
|
|
|
152
152
|
|
|
153
153
|
By combining all three with RRF fusion, MicroRAG achieves better recall and precision than any single method alone.
|
|
154
154
|
|
|
155
|
+
### Filtering Irrelevant Results
|
|
156
|
+
|
|
157
|
+
By default, search returns the top-k results regardless of relevance. For queries like "111111" or random gibberish, the system will still return documents (just with lower scores). To filter out irrelevant results, use `similarity_threshold`.
|
|
158
|
+
|
|
159
|
+
**Understanding RRF scores:** RRF fusion produces scores typically in the 0.01-0.03 range, not 0-1 like raw cosine similarity. This is because RRF scores are based on rank positions: `score = Σ weight / (k + rank)`.
|
|
160
|
+
|
|
161
|
+
**Finding the right threshold:**
|
|
162
|
+
|
|
163
|
+
```python
|
|
164
|
+
# Test with your data to find appropriate threshold
|
|
165
|
+
results = rag.search("relevant query", threshold=0.0)
|
|
166
|
+
print(f"Relevant score: {results[0].score}") # e.g., 0.016
|
|
167
|
+
|
|
168
|
+
results = rag.search("gibberish123", threshold=0.0)
|
|
169
|
+
print(f"Irrelevant score: {results[0].score}") # e.g., 0.011
|
|
170
|
+
|
|
171
|
+
# Set threshold between irrelevant and relevant scores
|
|
172
|
+
config = RAGConfig(
|
|
173
|
+
model_path="...",
|
|
174
|
+
similarity_threshold=0.014, # Filters gibberish, keeps relevant
|
|
175
|
+
)
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
**Typical thresholds:**
|
|
179
|
+
- `0.0` - Return all results (no filtering)
|
|
180
|
+
- `0.010-0.015` - Filter obvious gibberish while keeping most relevant results
|
|
181
|
+
- `0.015-0.020` - Stricter filtering, may reduce recall for edge cases
|
|
182
|
+
|
|
155
183
|
## Configuration
|
|
156
184
|
|
|
157
185
|
```python
|
|
@@ -173,7 +201,7 @@ config = RAGConfig(
|
|
|
173
201
|
# Search
|
|
174
202
|
hybrid_enabled=True, # Enable hybrid search
|
|
175
203
|
hybrid_alpha=0.7, # Semantic weight (0-1)
|
|
176
|
-
similarity_threshold=0.
|
|
204
|
+
similarity_threshold=0.014, # Min score threshold (RRF scores are ~0.01-0.03)
|
|
177
205
|
|
|
178
206
|
# Query processing
|
|
179
207
|
abbreviations={"ML": "machine learning"}, # Query expansion
|
|
@@ -205,7 +233,7 @@ config = RAGConfig(
|
|
|
205
233
|
**Search:**
|
|
206
234
|
- `hybrid_enabled` (bool, default: True) - Enable hybrid search
|
|
207
235
|
- `hybrid_alpha` (float, default: 0.7) - Semantic weight in fusion (0-1)
|
|
208
|
-
- `similarity_threshold` (float, default: 0.4) - Minimum score to return
|
|
236
|
+
- `similarity_threshold` (float, default: 0.4) - Minimum score to return (see [Filtering Irrelevant Results](#filtering-irrelevant-results))
|
|
209
237
|
|
|
210
238
|
**Query Processing:**
|
|
211
239
|
- `abbreviations` (dict, default: None) - Query expansion mapping
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|