microrag 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. {microrag-0.2.2 → microrag-0.2.4}/LICENSE +1 -1
  2. {microrag-0.2.2 → microrag-0.2.4}/PKG-INFO +32 -4
  3. {microrag-0.2.2 → microrag-0.2.4}/README.md +30 -2
  4. {microrag-0.2.2 → microrag-0.2.4}/pyproject.toml +1 -1
  5. {microrag-0.2.2 → microrag-0.2.4}/.github/workflows/ci.yml +0 -0
  6. {microrag-0.2.2 → microrag-0.2.4}/.github/workflows/publish.yml +0 -0
  7. {microrag-0.2.2 → microrag-0.2.4}/.gitignore +0 -0
  8. {microrag-0.2.2 → microrag-0.2.4}/Makefile +0 -0
  9. {microrag-0.2.2 → microrag-0.2.4}/examples/advanced_config.py +0 -0
  10. {microrag-0.2.2 → microrag-0.2.4}/examples/basic_usage.py +0 -0
  11. {microrag-0.2.2 → microrag-0.2.4}/examples/faq_search.py +0 -0
  12. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/__init__.py +0 -0
  13. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/config.py +0 -0
  14. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/core.py +0 -0
  15. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/__init__.py +0 -0
  16. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/base.py +0 -0
  17. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/factory.py +0 -0
  18. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/fastembed.py +0 -0
  19. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/embedding/sentence_transformers.py +0 -0
  20. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/exceptions.py +0 -0
  21. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/models.py +0 -0
  22. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/query_processor.py +0 -0
  23. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/search/__init__.py +0 -0
  24. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/search/bm25.py +0 -0
  25. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/search/hybrid.py +0 -0
  26. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/stopwords.py +0 -0
  27. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/storage/__init__.py +0 -0
  28. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/storage/base.py +0 -0
  29. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/storage/duckdb.py +0 -0
  30. {microrag-0.2.2 → microrag-0.2.4}/src/microrag/utils.py +0 -0
  31. {microrag-0.2.2 → microrag-0.2.4}/tests/__init__.py +0 -0
  32. {microrag-0.2.2 → microrag-0.2.4}/tests/conftest.py +0 -0
  33. {microrag-0.2.2 → microrag-0.2.4}/tests/test_bm25.py +0 -0
  34. {microrag-0.2.2 → microrag-0.2.4}/tests/test_config.py +0 -0
  35. {microrag-0.2.2 → microrag-0.2.4}/tests/test_integration.py +0 -0
  36. {microrag-0.2.2 → microrag-0.2.4}/tests/test_models.py +0 -0
  37. {microrag-0.2.2 → microrag-0.2.4}/tests/test_query_processor.py +0 -0
  38. {microrag-0.2.2 → microrag-0.2.4}/tests/test_storage.py +0 -0
  39. {microrag-0.2.2 → microrag-0.2.4}/tests/test_utils.py +0 -0
  40. {microrag-0.2.2 → microrag-0.2.4}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  MIT License
2
2
 
3
- Copyright (c) 2025 Dave Allie
3
+ Copyright (c) 2026 Pavel Liashkov
4
4
 
5
5
  Permission is hereby granted, free of charge, to any person obtaining a copy
6
6
  of this software and associated documentation files (the "Software"), to deal
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: microrag
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: A feature-rich, universal RAG library for Python with ONNX-backed embeddings and DuckDB storage
5
5
  Author-email: Pavel Liashkov <pavel.liashkov@protonamil.com>
6
6
  License-Expression: MIT
@@ -188,6 +188,34 @@ Each search method has different strengths:
188
188
 
189
189
  By combining all three with RRF fusion, MicroRAG achieves better recall and precision than any single method alone.
190
190
 
191
+ ### Filtering Irrelevant Results
192
+
193
+ By default, search returns the top-k results regardless of relevance. For queries like "111111" or random gibberish, the system will still return documents (just with lower scores). To filter out irrelevant results, use `similarity_threshold`.
194
+
195
+ **Understanding RRF scores:** RRF fusion produces scores typically in the 0.01-0.03 range, not 0-1 like raw cosine similarity. This is because RRF scores are based on rank positions: `score = Σ weight / (k + rank)`.
196
+
197
+ **Finding the right threshold:**
198
+
199
+ ```python
200
+ # Test with your data to find appropriate threshold
201
+ results = rag.search("relevant query", threshold=0.0)
202
+ print(f"Relevant score: {results[0].score}") # e.g., 0.016
203
+
204
+ results = rag.search("gibberish123", threshold=0.0)
205
+ print(f"Irrelevant score: {results[0].score}") # e.g., 0.011
206
+
207
+ # Set threshold between irrelevant and relevant scores
208
+ config = RAGConfig(
209
+ model_path="...",
210
+ similarity_threshold=0.014, # Filters gibberish, keeps relevant
211
+ )
212
+ ```
213
+
214
+ **Typical thresholds:**
215
+ - `0.0` - Return all results (no filtering)
216
+ - `0.010-0.015` - Filter obvious gibberish while keeping most relevant results
217
+ - `0.015-0.020` - Stricter filtering, may reduce recall for edge cases
218
+
191
219
  ## Configuration
192
220
 
193
221
  ```python
@@ -209,7 +237,7 @@ config = RAGConfig(
209
237
  # Search
210
238
  hybrid_enabled=True, # Enable hybrid search
211
239
  hybrid_alpha=0.7, # Semantic weight (0-1)
212
- similarity_threshold=0.4, # Min score threshold
240
+ similarity_threshold=0.014, # Min score threshold (RRF scores are ~0.01-0.03)
213
241
 
214
242
  # Query processing
215
243
  abbreviations={"ML": "machine learning"}, # Query expansion
@@ -241,7 +269,7 @@ config = RAGConfig(
241
269
  **Search:**
242
270
  - `hybrid_enabled` (bool, default: True) - Enable hybrid search
243
271
  - `hybrid_alpha` (float, default: 0.7) - Semantic weight in fusion (0-1)
244
- - `similarity_threshold` (float, default: 0.4) - Minimum score to return
272
+ - `similarity_threshold` (float, default: 0.4) - Minimum score to return (see [Filtering Irrelevant Results](#filtering-irrelevant-results))
245
273
 
246
274
  **Query Processing:**
247
275
  - `abbreviations` (dict, default: None) - Query expansion mapping
@@ -152,6 +152,34 @@ Each search method has different strengths:
152
152
 
153
153
  By combining all three with RRF fusion, MicroRAG achieves better recall and precision than any single method alone.
154
154
 
155
+ ### Filtering Irrelevant Results
156
+
157
+ By default, search returns the top-k results regardless of relevance. For queries like "111111" or random gibberish, the system will still return documents (just with lower scores). To filter out irrelevant results, use `similarity_threshold`.
158
+
159
+ **Understanding RRF scores:** RRF fusion produces scores typically in the 0.01-0.03 range, not 0-1 like raw cosine similarity. This is because RRF scores are based on rank positions: `score = Σ weight / (k + rank)`.
160
+
161
+ **Finding the right threshold:**
162
+
163
+ ```python
164
+ # Test with your data to find appropriate threshold
165
+ results = rag.search("relevant query", threshold=0.0)
166
+ print(f"Relevant score: {results[0].score}") # e.g., 0.016
167
+
168
+ results = rag.search("gibberish123", threshold=0.0)
169
+ print(f"Irrelevant score: {results[0].score}") # e.g., 0.011
170
+
171
+ # Set threshold between irrelevant and relevant scores
172
+ config = RAGConfig(
173
+ model_path="...",
174
+ similarity_threshold=0.014, # Filters gibberish, keeps relevant
175
+ )
176
+ ```
177
+
178
+ **Typical thresholds:**
179
+ - `0.0` - Return all results (no filtering)
180
+ - `0.010-0.015` - Filter obvious gibberish while keeping most relevant results
181
+ - `0.015-0.020` - Stricter filtering, may reduce recall for edge cases
182
+
155
183
  ## Configuration
156
184
 
157
185
  ```python
@@ -173,7 +201,7 @@ config = RAGConfig(
173
201
  # Search
174
202
  hybrid_enabled=True, # Enable hybrid search
175
203
  hybrid_alpha=0.7, # Semantic weight (0-1)
176
- similarity_threshold=0.4, # Min score threshold
204
+ similarity_threshold=0.014, # Min score threshold (RRF scores are ~0.01-0.03)
177
205
 
178
206
  # Query processing
179
207
  abbreviations={"ML": "machine learning"}, # Query expansion
@@ -205,7 +233,7 @@ config = RAGConfig(
205
233
  **Search:**
206
234
  - `hybrid_enabled` (bool, default: True) - Enable hybrid search
207
235
  - `hybrid_alpha` (float, default: 0.7) - Semantic weight in fusion (0-1)
208
- - `similarity_threshold` (float, default: 0.4) - Minimum score to return
236
+ - `similarity_threshold` (float, default: 0.4) - Minimum score to return (see [Filtering Irrelevant Results](#filtering-irrelevant-results))
209
237
 
210
238
  **Query Processing:**
211
239
  - `abbreviations` (dict, default: None) - Query expansion mapping
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "microrag"
3
- version = "0.2.2"
3
+ version = "0.2.4"
4
4
  description = "A feature-rich, universal RAG library for Python with ONNX-backed embeddings and DuckDB storage"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes