qdrant-haystack 11.0.0__tar.gz → 12.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/CHANGELOG.md +22 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/PKG-INFO +1 -1
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/components/retrievers/qdrant/retriever.py +30 -27
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/document_stores/qdrant/document_store.py +9 -8
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/document_stores/qdrant/filters.py +7 -31
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_embedding_retriever.py +31 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_filters.py +52 -10
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_hybrid_retriever.py +33 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_sparse_embedding_retriever.py +33 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/.gitignore +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/LICENSE.txt +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/README.md +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/examples/embedding_retrieval.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/pydoc/config_docusaurus.yml +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/pyproject.toml +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/components/retrievers/py.typed +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/components/retrievers/qdrant/__init__.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/document_stores/py.typed +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/document_stores/qdrant/__init__.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/document_stores/qdrant/converters.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/document_stores/qdrant/migrate_to_sparse.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/__init__.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/conftest.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_converters.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_dict_converters.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_document_store.py +0 -0
- {qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/tests/test_document_store_async.py +0 -0
|
@@ -1,5 +1,27 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [integrations/qdrant-v11.1.0] - 2026-10-01
|
|
4
|
+
|
|
5
|
+
### 🐛 Bug Fixes
|
|
6
|
+
|
|
7
|
+
- *(qdrant)* Respect falsy run() overrides in retrievers (#4007)
|
|
8
|
+
|
|
9
|
+
### 📚 Documentation
|
|
10
|
+
|
|
11
|
+
- Clarify score_threshold semantics for Qdrant hybrid retrieval (#4008)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
## [integrations/qdrant-v11.0.0] - 2026-09-25
|
|
15
|
+
|
|
16
|
+
### 🚜 Refactor
|
|
17
|
+
|
|
18
|
+
- [**breaking**] Qdrant - propagate errors on backend failures instead of empty responses (#3997)
|
|
19
|
+
|
|
20
|
+
### ⚙️ CI
|
|
21
|
+
|
|
22
|
+
- Improve changelog generation; fix existing changelogs (#3883)
|
|
23
|
+
|
|
24
|
+
|
|
3
25
|
## [integrations/qdrant-v10.6.0] - 2026-09-01
|
|
4
26
|
|
|
5
27
|
### 🐛 Bug Fixes
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: qdrant-haystack
|
|
3
|
-
Version:
|
|
3
|
+
Version: 12.0.0
|
|
4
4
|
Summary: An integration of Qdrant ANN vector database backend with Haystack
|
|
5
5
|
Project-URL: Source, https://github.com/deepset-ai/haystack-core-integrations
|
|
6
6
|
Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/blob/main/integrations/qdrant/README.md
|
|
@@ -188,9 +188,9 @@ class QdrantEmbeddingRetriever:
|
|
|
188
188
|
query_embedding=query_embedding,
|
|
189
189
|
filters=filters,
|
|
190
190
|
top_k=top_k or self._top_k,
|
|
191
|
-
scale_score=scale_score
|
|
192
|
-
return_embedding=return_embedding
|
|
193
|
-
score_threshold=score_threshold
|
|
191
|
+
scale_score=scale_score if scale_score is not None else self._scale_score,
|
|
192
|
+
return_embedding=return_embedding if return_embedding is not None else self._return_embedding,
|
|
193
|
+
score_threshold=score_threshold if score_threshold is not None else self._score_threshold,
|
|
194
194
|
group_by=group_by or self._group_by,
|
|
195
195
|
group_size=group_size or self._group_size,
|
|
196
196
|
)
|
|
@@ -243,9 +243,9 @@ class QdrantEmbeddingRetriever:
|
|
|
243
243
|
query_embedding=query_embedding,
|
|
244
244
|
filters=filters,
|
|
245
245
|
top_k=top_k or self._top_k,
|
|
246
|
-
scale_score=scale_score
|
|
247
|
-
return_embedding=return_embedding
|
|
248
|
-
score_threshold=score_threshold
|
|
246
|
+
scale_score=scale_score if scale_score is not None else self._scale_score,
|
|
247
|
+
return_embedding=return_embedding if return_embedding is not None else self._return_embedding,
|
|
248
|
+
score_threshold=score_threshold if score_threshold is not None else self._score_threshold,
|
|
249
249
|
group_by=group_by or self._group_by,
|
|
250
250
|
group_size=group_size or self._group_size,
|
|
251
251
|
)
|
|
@@ -433,9 +433,9 @@ class QdrantSparseEmbeddingRetriever:
|
|
|
433
433
|
query_sparse_embedding=query_sparse_embedding,
|
|
434
434
|
filters=filters,
|
|
435
435
|
top_k=top_k or self._top_k,
|
|
436
|
-
scale_score=scale_score
|
|
437
|
-
return_embedding=return_embedding
|
|
438
|
-
score_threshold=score_threshold
|
|
436
|
+
scale_score=scale_score if scale_score is not None else self._scale_score,
|
|
437
|
+
return_embedding=return_embedding if return_embedding is not None else self._return_embedding,
|
|
438
|
+
score_threshold=score_threshold if score_threshold is not None else self._score_threshold,
|
|
439
439
|
group_by=group_by or self._group_by,
|
|
440
440
|
group_size=group_size or self._group_size,
|
|
441
441
|
)
|
|
@@ -493,9 +493,9 @@ class QdrantSparseEmbeddingRetriever:
|
|
|
493
493
|
query_sparse_embedding=query_sparse_embedding,
|
|
494
494
|
filters=filters,
|
|
495
495
|
top_k=top_k or self._top_k,
|
|
496
|
-
scale_score=scale_score
|
|
497
|
-
return_embedding=return_embedding
|
|
498
|
-
score_threshold=score_threshold
|
|
496
|
+
scale_score=scale_score if scale_score is not None else self._scale_score,
|
|
497
|
+
return_embedding=return_embedding if return_embedding is not None else self._return_embedding,
|
|
498
|
+
score_threshold=score_threshold if score_threshold is not None else self._score_threshold,
|
|
499
499
|
group_by=group_by or self._group_by,
|
|
500
500
|
group_size=group_size or self._group_size,
|
|
501
501
|
)
|
|
@@ -560,9 +560,10 @@ class QdrantHybridRetriever:
|
|
|
560
560
|
:param return_embedding: Whether to return the embeddings of the retrieved Documents.
|
|
561
561
|
:param filter_policy: Policy to determine how filters are applied.
|
|
562
562
|
:param score_threshold: A minimal score threshold for the result.
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
563
|
+
The threshold is applied to the fused Reciprocal Rank Fusion (RRF) scores, so only documents with a
|
|
564
|
+
higher fused score are returned. RRF scores are not comparable to dense or sparse similarity scores,
|
|
565
|
+
so thresholds tuned for those retrievers don't carry over. With `group_by`, groups with no surviving
|
|
566
|
+
hits are dropped.
|
|
566
567
|
:param group_by: Payload field to group by, must be a string or number field. If the field contains more than 1
|
|
567
568
|
value, all values will be used for grouping. One point can be in multiple groups.
|
|
568
569
|
:param group_size: Maximum amount of points to return per group. Default is 3.
|
|
@@ -659,7 +660,7 @@ class QdrantHybridRetriever:
|
|
|
659
660
|
rrf_weights: list[float] | None = None,
|
|
660
661
|
) -> dict[str, list[Document]]:
|
|
661
662
|
"""
|
|
662
|
-
Run the
|
|
663
|
+
Run the Hybrid Retriever on the given input data.
|
|
663
664
|
|
|
664
665
|
:param query_embedding: Dense embedding of the query.
|
|
665
666
|
:param query_sparse_embedding: Sparse embedding of the query.
|
|
@@ -670,9 +671,10 @@ class QdrantHybridRetriever:
|
|
|
670
671
|
groups to return.
|
|
671
672
|
:param return_embedding: Whether to return the embedding of the retrieved Documents.
|
|
672
673
|
:param score_threshold: A minimal score threshold for the result.
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
674
|
+
The threshold is applied to the fused Reciprocal Rank Fusion (RRF) scores, so only documents with a
|
|
675
|
+
higher fused score are returned. RRF scores are not comparable to dense or sparse similarity scores,
|
|
676
|
+
so thresholds tuned for those retrievers don't carry over. With `group_by`, groups with no surviving
|
|
677
|
+
hits are dropped.
|
|
676
678
|
:param group_by: Payload field to group by, must be a string or number field. If the field contains more than 1
|
|
677
679
|
value, all values will be used for grouping. One point can be in multiple groups.
|
|
678
680
|
:param group_size: Maximum amount of points to return per group. Default is 3.
|
|
@@ -704,8 +706,8 @@ class QdrantHybridRetriever:
|
|
|
704
706
|
query_sparse_embedding=query_sparse_embedding,
|
|
705
707
|
filters=filters,
|
|
706
708
|
top_k=top_k or self._top_k,
|
|
707
|
-
return_embedding=return_embedding
|
|
708
|
-
score_threshold=score_threshold
|
|
709
|
+
return_embedding=return_embedding if return_embedding is not None else self._return_embedding,
|
|
710
|
+
score_threshold=score_threshold if score_threshold is not None else self._score_threshold,
|
|
709
711
|
group_by=group_by or self._group_by,
|
|
710
712
|
group_size=group_size or self._group_size,
|
|
711
713
|
rrf_k=rrf_k if rrf_k is not None else self._rrf_k,
|
|
@@ -729,7 +731,7 @@ class QdrantHybridRetriever:
|
|
|
729
731
|
rrf_weights: list[float] | None = None,
|
|
730
732
|
) -> dict[str, list[Document]]:
|
|
731
733
|
"""
|
|
732
|
-
Asynchronously run the
|
|
734
|
+
Asynchronously run the Hybrid Retriever on the given input data.
|
|
733
735
|
|
|
734
736
|
:param query_embedding: Dense embedding of the query.
|
|
735
737
|
:param query_sparse_embedding: Sparse embedding of the query.
|
|
@@ -740,9 +742,10 @@ class QdrantHybridRetriever:
|
|
|
740
742
|
groups to return.
|
|
741
743
|
:param return_embedding: Whether to return the embedding of the retrieved Documents.
|
|
742
744
|
:param score_threshold: A minimal score threshold for the result.
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
745
|
+
The threshold is applied to the fused Reciprocal Rank Fusion (RRF) scores, so only documents with a
|
|
746
|
+
higher fused score are returned. RRF scores are not comparable to dense or sparse similarity scores,
|
|
747
|
+
so thresholds tuned for those retrievers don't carry over. With `group_by`, groups with no surviving
|
|
748
|
+
hits are dropped.
|
|
746
749
|
:param group_by: Payload field to group by, must be a string or number field. If the field contains more than 1
|
|
747
750
|
value, all values will be used for grouping. One point can be in multiple groups.
|
|
748
751
|
:param group_size: Maximum amount of points to return per group. Default is 3.
|
|
@@ -774,8 +777,8 @@ class QdrantHybridRetriever:
|
|
|
774
777
|
query_sparse_embedding=query_sparse_embedding,
|
|
775
778
|
filters=filters,
|
|
776
779
|
top_k=top_k or self._top_k,
|
|
777
|
-
return_embedding=return_embedding
|
|
778
|
-
score_threshold=score_threshold
|
|
780
|
+
return_embedding=return_embedding if return_embedding is not None else self._return_embedding,
|
|
781
|
+
score_threshold=score_threshold if score_threshold is not None else self._score_threshold,
|
|
779
782
|
group_by=group_by or self._group_by,
|
|
780
783
|
group_size=group_size or self._group_size,
|
|
781
784
|
rrf_k=rrf_k if rrf_k is not None else self._rrf_k,
|
|
@@ -11,7 +11,7 @@ from haystack.dataclasses import Document
|
|
|
11
11
|
from haystack.dataclasses.sparse_embedding import SparseEmbedding
|
|
12
12
|
from haystack.document_stores.errors import DocumentStoreError, DuplicateDocumentError
|
|
13
13
|
from haystack.document_stores.types import DuplicatePolicy
|
|
14
|
-
from haystack.utils import Secret
|
|
14
|
+
from haystack.utils import Secret
|
|
15
15
|
from haystack.utils.misc import _normalize_metadata_field_name
|
|
16
16
|
from numpy import exp
|
|
17
17
|
from qdrant_client.http import models as rest
|
|
@@ -1432,7 +1432,6 @@ class QdrantDocumentStore:
|
|
|
1432
1432
|
:returns:
|
|
1433
1433
|
The deserialized component.
|
|
1434
1434
|
"""
|
|
1435
|
-
deserialize_secrets_inplace(data["init_parameters"], keys=["api_key"])
|
|
1436
1435
|
return default_from_dict(cls, data)
|
|
1437
1436
|
|
|
1438
1437
|
def to_dict(self) -> dict[str, Any]:
|
|
@@ -1757,9 +1756,10 @@ class QdrantDocumentStore:
|
|
|
1757
1756
|
groups to return.
|
|
1758
1757
|
:param return_embedding: Whether to return the embeddings of the retrieved documents.
|
|
1759
1758
|
:param score_threshold: A minimal score threshold for the result.
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1759
|
+
The threshold is applied to the fused Reciprocal Rank Fusion (RRF) scores, so only documents with a
|
|
1760
|
+
higher fused score are returned. RRF scores are not comparable to dense or sparse similarity scores,
|
|
1761
|
+
so thresholds tuned for those retrievers don't carry over. With `group_by`, groups with no surviving
|
|
1762
|
+
hits are dropped.
|
|
1763
1763
|
:param group_by: Payload field to group by, must be a string or number field. If the field contains more than 1
|
|
1764
1764
|
value, all values will be used for grouping. One point can be in multiple groups.
|
|
1765
1765
|
:param group_size: Maximum amount of points to return per group. Default is 3.
|
|
@@ -2020,9 +2020,10 @@ class QdrantDocumentStore:
|
|
|
2020
2020
|
groups to return.
|
|
2021
2021
|
:param return_embedding: Whether to return the embeddings of the retrieved documents.
|
|
2022
2022
|
:param score_threshold: A minimal score threshold for the result.
|
|
2023
|
-
|
|
2024
|
-
|
|
2025
|
-
|
|
2023
|
+
The threshold is applied to the fused Reciprocal Rank Fusion (RRF) scores, so only documents with a
|
|
2024
|
+
higher fused score are returned. RRF scores are not comparable to dense or sparse similarity scores,
|
|
2025
|
+
so thresholds tuned for those retrievers don't carry over. With `group_by`, groups with no surviving
|
|
2026
|
+
hits are dropped.
|
|
2026
2027
|
:param group_by: Payload field to group by, must be a string or number field. If the field contains more than 1
|
|
2027
2028
|
value, all values will be used for grouping. One point can be in multiple groups.
|
|
2028
2029
|
:param group_size: Maximum amount of points to return per group. Default is 3.
|
|
@@ -1,3 +1,7 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: 2023-present deepset GmbH <info@deepset.ai>
|
|
2
|
+
#
|
|
3
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
|
|
1
5
|
from collections.abc import Callable
|
|
2
6
|
from datetime import datetime
|
|
3
7
|
from typing import Any
|
|
@@ -131,8 +135,6 @@ def _build_comparison_condition(operator: str, key: str, value: Any) -> models.C
|
|
|
131
135
|
|
|
132
136
|
|
|
133
137
|
def _build_eq_condition(key: str, value: models.ValueVariants) -> models.Condition:
|
|
134
|
-
if isinstance(value, str) and " " in value:
|
|
135
|
-
return models.FieldCondition(key=key, match=models.MatchText(text=value))
|
|
136
138
|
return models.FieldCondition(key=key, match=models.MatchValue(value=value))
|
|
137
139
|
|
|
138
140
|
|
|
@@ -140,44 +142,18 @@ def _build_in_condition(key: str, value: list[models.ValueVariants]) -> models.C
|
|
|
140
142
|
if not isinstance(value, list):
|
|
141
143
|
msg = f"Value {value} is not a list"
|
|
142
144
|
raise FilterError(msg)
|
|
143
|
-
return models.Filter(
|
|
144
|
-
should=[
|
|
145
|
-
(
|
|
146
|
-
models.FieldCondition(key=key, match=models.MatchText(text=item))
|
|
147
|
-
if isinstance(item, str) and " " not in item
|
|
148
|
-
else models.FieldCondition(key=key, match=models.MatchValue(value=item))
|
|
149
|
-
)
|
|
150
|
-
for item in value
|
|
151
|
-
]
|
|
152
|
-
)
|
|
145
|
+
return models.Filter(should=[_build_eq_condition(key, item) for item in value])
|
|
153
146
|
|
|
154
147
|
|
|
155
148
|
def _build_ne_condition(key: str, value: models.ValueVariants) -> models.Condition:
|
|
156
|
-
return models.Filter(
|
|
157
|
-
must_not=[
|
|
158
|
-
(
|
|
159
|
-
models.FieldCondition(key=key, match=models.MatchText(text=value))
|
|
160
|
-
if isinstance(value, str) and " " not in value
|
|
161
|
-
else models.FieldCondition(key=key, match=models.MatchValue(value=value))
|
|
162
|
-
)
|
|
163
|
-
]
|
|
164
|
-
)
|
|
149
|
+
return models.Filter(must_not=[_build_eq_condition(key, value)])
|
|
165
150
|
|
|
166
151
|
|
|
167
152
|
def _build_nin_condition(key: str, value: list[models.ValueVariants]) -> models.Condition:
|
|
168
153
|
if not isinstance(value, list):
|
|
169
154
|
msg = f"Value {value} is not a list"
|
|
170
155
|
raise FilterError(msg)
|
|
171
|
-
return models.Filter(
|
|
172
|
-
must_not=[
|
|
173
|
-
(
|
|
174
|
-
models.FieldCondition(key=key, match=models.MatchText(text=item))
|
|
175
|
-
if isinstance(item, str) and " " in item
|
|
176
|
-
else models.FieldCondition(key=key, match=models.MatchValue(value=item))
|
|
177
|
-
)
|
|
178
|
-
for item in value
|
|
179
|
-
]
|
|
180
|
-
)
|
|
156
|
+
return models.Filter(must_not=[_build_eq_condition(key, item) for item in value])
|
|
181
157
|
|
|
182
158
|
|
|
183
159
|
def _build_lt_condition(key: str, value: str | float | int) -> models.Condition:
|
|
@@ -200,6 +200,37 @@ class TestQdrantRetriever:
|
|
|
200
200
|
mock_store._query_by_embedding_async.assert_awaited_once()
|
|
201
201
|
assert res["documents"][0].content == "doc"
|
|
202
202
|
|
|
203
|
+
def test_run_falsy_runtime_values_override_init(self):
|
|
204
|
+
mock_store = Mock(spec=QdrantDocumentStore)
|
|
205
|
+
mock_store._query_by_embedding.return_value = []
|
|
206
|
+
|
|
207
|
+
retriever = QdrantEmbeddingRetriever(
|
|
208
|
+
document_store=mock_store, scale_score=True, return_embedding=True, score_threshold=0.5
|
|
209
|
+
)
|
|
210
|
+
retriever.run(query_embedding=[0.5, 0.7], scale_score=False, return_embedding=False, score_threshold=0.0)
|
|
211
|
+
|
|
212
|
+
call_kwargs = mock_store._query_by_embedding.call_args.kwargs
|
|
213
|
+
assert call_kwargs["scale_score"] is False
|
|
214
|
+
assert call_kwargs["return_embedding"] is False
|
|
215
|
+
assert call_kwargs["score_threshold"] == 0.0
|
|
216
|
+
|
|
217
|
+
@pytest.mark.asyncio
|
|
218
|
+
async def test_run_async_falsy_runtime_values_override_init(self):
|
|
219
|
+
mock_store = Mock(spec=QdrantDocumentStore)
|
|
220
|
+
mock_store._query_by_embedding_async = AsyncMock(return_value=[])
|
|
221
|
+
|
|
222
|
+
retriever = QdrantEmbeddingRetriever(
|
|
223
|
+
document_store=mock_store, scale_score=True, return_embedding=True, score_threshold=0.5
|
|
224
|
+
)
|
|
225
|
+
await retriever.run_async(
|
|
226
|
+
query_embedding=[0.5, 0.7], scale_score=False, return_embedding=False, score_threshold=0.0
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
call_kwargs = mock_store._query_by_embedding_async.call_args.kwargs
|
|
230
|
+
assert call_kwargs["scale_score"] is False
|
|
231
|
+
assert call_kwargs["return_embedding"] is False
|
|
232
|
+
assert call_kwargs["score_threshold"] == 0.0
|
|
233
|
+
|
|
203
234
|
def test_run_raises_when_merge_with_native_init_filter(self):
|
|
204
235
|
document_store = QdrantDocumentStore(location=":memory:", index="test")
|
|
205
236
|
retriever = QdrantEmbeddingRetriever(
|
|
@@ -1,3 +1,7 @@
|
|
|
1
|
+
# SPDX-FileCopyrightText: 2023-present deepset GmbH <info@deepset.ai>
|
|
2
|
+
#
|
|
3
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
|
|
1
5
|
import pytest
|
|
2
6
|
from haystack import Document
|
|
3
7
|
from haystack.testing.document_store import FilterDocumentsTest
|
|
@@ -49,16 +53,6 @@ class TestConvertFiltersToQdrantUnit:
|
|
|
49
53
|
)
|
|
50
54
|
assert isinstance(qdrant_filter, models.Filter)
|
|
51
55
|
|
|
52
|
-
def test_eq_with_spaces_uses_text_match(self):
|
|
53
|
-
qdrant_filter = convert_filters_to_qdrant({"operator": "==", "field": "meta.title", "value": "hello world"})
|
|
54
|
-
condition = qdrant_filter.must[0]
|
|
55
|
-
assert isinstance(condition.match, models.MatchText)
|
|
56
|
-
|
|
57
|
-
def test_eq_without_spaces_uses_value_match(self):
|
|
58
|
-
qdrant_filter = convert_filters_to_qdrant({"operator": "==", "field": "meta.name", "value": "name_0"})
|
|
59
|
-
condition = qdrant_filter.must[0]
|
|
60
|
-
assert isinstance(condition.match, models.MatchValue)
|
|
61
|
-
|
|
62
56
|
def test_single_logical_condition_unwrapped(self):
|
|
63
57
|
qdrant_filter = convert_filters_to_qdrant(
|
|
64
58
|
{
|
|
@@ -96,6 +90,54 @@ class TestQdrantFilters(FilterDocumentsTest):
|
|
|
96
90
|
wait_result_from_api=True,
|
|
97
91
|
)
|
|
98
92
|
|
|
93
|
+
@pytest.mark.parametrize(("value", "longer_value"), [("news", "newsletter"), ("New York", "New York City")])
|
|
94
|
+
def test_eq_matches_exact_string(self, document_store, value, longer_value):
|
|
95
|
+
exact_document = Document(content=value, meta={"name": value})
|
|
96
|
+
other_document = Document(content=longer_value, meta={"name": longer_value})
|
|
97
|
+
document_store.write_documents([exact_document, other_document])
|
|
98
|
+
|
|
99
|
+
result = document_store.filter_documents(filters={"field": "meta.name", "operator": "==", "value": value})
|
|
100
|
+
|
|
101
|
+
self.assert_documents_are_equal(result, [exact_document])
|
|
102
|
+
|
|
103
|
+
@pytest.mark.parametrize(("value", "longer_value"), [("news", "newsletter"), ("New York", "New York City")])
|
|
104
|
+
def test_ne_excludes_exact_string(self, document_store, value, longer_value):
|
|
105
|
+
exact_document = Document(content=value, meta={"name": value})
|
|
106
|
+
other_document = Document(content=longer_value, meta={"name": longer_value})
|
|
107
|
+
document_store.write_documents([exact_document, other_document])
|
|
108
|
+
|
|
109
|
+
result = document_store.filter_documents(filters={"field": "meta.name", "operator": "!=", "value": value})
|
|
110
|
+
|
|
111
|
+
self.assert_documents_are_equal(result, [other_document])
|
|
112
|
+
|
|
113
|
+
def test_in_matches_exact_strings(self, document_store):
|
|
114
|
+
document_store.write_documents(
|
|
115
|
+
[
|
|
116
|
+
Document(content=name, meta={"name": name})
|
|
117
|
+
for name in ["news", "newsletter", "New York", "New York City"]
|
|
118
|
+
]
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
result = document_store.filter_documents(
|
|
122
|
+
filters={"field": "meta.name", "operator": "in", "value": ["news", "New York"]}
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
assert sorted(doc.meta["name"] for doc in result) == ["New York", "news"]
|
|
126
|
+
|
|
127
|
+
def test_not_in_excludes_exact_strings(self, document_store):
|
|
128
|
+
document_store.write_documents(
|
|
129
|
+
[
|
|
130
|
+
Document(content=name, meta={"name": name})
|
|
131
|
+
for name in ["news", "newsletter", "New York", "New York City"]
|
|
132
|
+
]
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
result = document_store.filter_documents(
|
|
136
|
+
filters={"field": "meta.name", "operator": "not in", "value": ["news", "New York"]}
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
assert sorted(doc.meta["name"] for doc in result) == ["New York City", "newsletter"]
|
|
140
|
+
|
|
99
141
|
def test_filter_documents_with_qdrant_filters(self, document_store, filterable_docs):
|
|
100
142
|
document_store.write_documents(filterable_docs)
|
|
101
143
|
result = document_store.filter_documents(
|
|
@@ -229,6 +229,39 @@ class TestQdrantHybridRetriever:
|
|
|
229
229
|
assert call_args[1]["rrf_k"] == 100
|
|
230
230
|
assert call_args[1]["rrf_weights"] == [3.0, 1.0]
|
|
231
231
|
|
|
232
|
+
def test_run_falsy_runtime_values_override_init(self):
|
|
233
|
+
mock_store = Mock(spec=QdrantDocumentStore)
|
|
234
|
+
mock_store._query_hybrid.return_value = []
|
|
235
|
+
|
|
236
|
+
retriever = QdrantHybridRetriever(document_store=mock_store, return_embedding=True, score_threshold=0.5)
|
|
237
|
+
retriever.run(
|
|
238
|
+
query_embedding=[0.5, 0.7],
|
|
239
|
+
query_sparse_embedding=SparseEmbedding(indices=[0, 5], values=[0.1, 0.7]),
|
|
240
|
+
return_embedding=False,
|
|
241
|
+
score_threshold=0.0,
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
call_args = mock_store._query_hybrid.call_args
|
|
245
|
+
assert call_args[1]["return_embedding"] is False
|
|
246
|
+
assert call_args[1]["score_threshold"] == 0.0
|
|
247
|
+
|
|
248
|
+
@pytest.mark.asyncio
|
|
249
|
+
async def test_run_async_falsy_runtime_values_override_init(self):
|
|
250
|
+
mock_store = Mock(spec=QdrantDocumentStore)
|
|
251
|
+
mock_store._query_hybrid_async = AsyncMock(return_value=[])
|
|
252
|
+
|
|
253
|
+
retriever = QdrantHybridRetriever(document_store=mock_store, return_embedding=True, score_threshold=0.5)
|
|
254
|
+
await retriever.run_async(
|
|
255
|
+
query_embedding=[0.5, 0.7],
|
|
256
|
+
query_sparse_embedding=SparseEmbedding(indices=[0, 5], values=[0.1, 0.7]),
|
|
257
|
+
return_embedding=False,
|
|
258
|
+
score_threshold=0.0,
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
call_args = mock_store._query_hybrid_async.call_args
|
|
262
|
+
assert call_args[1]["return_embedding"] is False
|
|
263
|
+
assert call_args[1]["score_threshold"] == 0.0
|
|
264
|
+
|
|
232
265
|
def test_run_with_group_by(self):
|
|
233
266
|
mock_store = Mock(spec=QdrantDocumentStore)
|
|
234
267
|
sparse_embedding = SparseEmbedding(indices=[0, 1, 2, 3], values=[0.1, 0.8, 0.05, 0.33])
|
|
@@ -199,6 +199,39 @@ class TestQdrantSparseEmbeddingRetriever:
|
|
|
199
199
|
mock_store._query_by_sparse_async.assert_awaited_once()
|
|
200
200
|
assert res["documents"][0].content == "doc"
|
|
201
201
|
|
|
202
|
+
def test_run_falsy_runtime_values_override_init(self):
|
|
203
|
+
mock_store = Mock(spec=QdrantDocumentStore)
|
|
204
|
+
mock_store._query_by_sparse.return_value = []
|
|
205
|
+
sparse = SparseEmbedding(indices=[0, 5], values=[0.1, 0.7])
|
|
206
|
+
|
|
207
|
+
retriever = QdrantSparseEmbeddingRetriever(
|
|
208
|
+
document_store=mock_store, scale_score=True, return_embedding=True, score_threshold=0.5
|
|
209
|
+
)
|
|
210
|
+
retriever.run(query_sparse_embedding=sparse, scale_score=False, return_embedding=False, score_threshold=0.0)
|
|
211
|
+
|
|
212
|
+
call_kwargs = mock_store._query_by_sparse.call_args.kwargs
|
|
213
|
+
assert call_kwargs["scale_score"] is False
|
|
214
|
+
assert call_kwargs["return_embedding"] is False
|
|
215
|
+
assert call_kwargs["score_threshold"] == 0.0
|
|
216
|
+
|
|
217
|
+
@pytest.mark.asyncio
|
|
218
|
+
async def test_run_async_falsy_runtime_values_override_init(self):
|
|
219
|
+
mock_store = Mock(spec=QdrantDocumentStore)
|
|
220
|
+
mock_store._query_by_sparse_async = AsyncMock(return_value=[])
|
|
221
|
+
sparse = SparseEmbedding(indices=[0, 5], values=[0.1, 0.7])
|
|
222
|
+
|
|
223
|
+
retriever = QdrantSparseEmbeddingRetriever(
|
|
224
|
+
document_store=mock_store, scale_score=True, return_embedding=True, score_threshold=0.5
|
|
225
|
+
)
|
|
226
|
+
await retriever.run_async(
|
|
227
|
+
query_sparse_embedding=sparse, scale_score=False, return_embedding=False, score_threshold=0.0
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
call_kwargs = mock_store._query_by_sparse_async.call_args.kwargs
|
|
231
|
+
assert call_kwargs["scale_score"] is False
|
|
232
|
+
assert call_kwargs["return_embedding"] is False
|
|
233
|
+
assert call_kwargs["score_threshold"] == 0.0
|
|
234
|
+
|
|
202
235
|
def test_run_raises_when_merge_with_native_filter(self):
|
|
203
236
|
document_store = QdrantDocumentStore(location=":memory:", index="test")
|
|
204
237
|
retriever = QdrantSparseEmbeddingRetriever(
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{qdrant_haystack-11.0.0 → qdrant_haystack-12.0.0}/src/haystack_integrations/document_stores/py.typed
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|