qdrant-haystack 10.3.0__tar.gz → 10.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/CHANGELOG.md +43 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/PKG-INFO +2 -2
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/examples/embedding_retrieval.py +4 -1
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/pyproject.toml +21 -4
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/components/retrievers/qdrant/retriever.py +39 -2
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/document_stores/qdrant/converters.py +2 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/document_stores/qdrant/document_store.py +48 -20
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/document_stores/qdrant/filters.py +3 -1
- qdrant_haystack-10.4.0/tests/test_converters.py +144 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/test_document_store.py +356 -130
- qdrant_haystack-10.4.0/tests/test_document_store_async.py +344 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/test_embedding_retriever.py +121 -5
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/test_filters.py +80 -1
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/test_hybrid_retriever.py +47 -1
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/test_sparse_embedding_retriever.py +94 -11
- qdrant_haystack-10.3.0/tests/test_converters.py +0 -62
- qdrant_haystack-10.3.0/tests/test_document_store_async.py +0 -640
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/.gitignore +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/LICENSE.txt +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/README.md +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/pydoc/config_docusaurus.yml +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/components/retrievers/py.typed +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/components/retrievers/qdrant/__init__.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/document_stores/py.typed +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/document_stores/qdrant/__init__.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/src/haystack_integrations/document_stores/qdrant/migrate_to_sparse.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/__init__.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/conftest.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.4.0}/tests/test_dict_converters.py +0 -0
|
@@ -1,5 +1,48 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [integrations/qdrant-v10.3.1] - 2026-07-06
|
|
4
|
+
|
|
5
|
+
### 🐛 Bug Fixes
|
|
6
|
+
|
|
7
|
+
- Replace in-place dataclass mutations in document stores (#3114)
|
|
8
|
+
|
|
9
|
+
### 📚 Documentation
|
|
10
|
+
|
|
11
|
+
- Replace old haystack core imports with haystack_integrations paths (#3545)
|
|
12
|
+
|
|
13
|
+
### 🧪 Testing
|
|
14
|
+
|
|
15
|
+
- Track test coverage for all integrations (#3065)
|
|
16
|
+
- Mark some tests as integration in Faiss and Qdrant (#3076)
|
|
17
|
+
- Qdrant - add unit tests (#3183)
|
|
18
|
+
- Qdrant - add more unit tests (#3221)
|
|
19
|
+
- *(qdrant)* Use async DocumentStore mixin tests (#3093)
|
|
20
|
+
|
|
21
|
+
### 🧹 Chores
|
|
22
|
+
|
|
23
|
+
- Enforce ruff docstring rules in integrations 31-40 (openrouter, opensearch, optimum, paddleocr, pgvector, pinecone, pyversity, qdrant, ragas, snowflake) (#3011)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
## [integrations/qdrant-v10.3.0] - 2026-03-23
|
|
27
|
+
|
|
28
|
+
### 📚 Documentation
|
|
29
|
+
|
|
30
|
+
- Simplify pydoc configs (#2855)
|
|
31
|
+
|
|
32
|
+
### 🧪 Testing
|
|
33
|
+
|
|
34
|
+
- Replacing each `DocumentStore` specific tests and used the generalised ones from `haystack.testing.document_store` (#2812)
|
|
35
|
+
- Test compatible integrations with python 3.14; update pyproject (#3001)
|
|
36
|
+
|
|
37
|
+
### 🧹 Chores
|
|
38
|
+
|
|
39
|
+
- Standardize author mentions (#2897)
|
|
40
|
+
- Add ANN ruff ruleset to optimum, paddleocr, pgvector, pinecone, pyversity, qdrant, ragas, snowflake (#2992)
|
|
41
|
+
|
|
42
|
+
### 🌀 Miscellaneous
|
|
43
|
+
|
|
44
|
+
- !test: `QdrantDocumentStore` use Mixin tests + updated signature `get_metadata_fields_info(self) -> dict[str, dict[str, str]]`: (#3004)
|
|
45
|
+
|
|
3
46
|
## [integrations/qdrant-v10.2.1] - 2026-02-02
|
|
4
47
|
|
|
5
48
|
### 📚 Documentation
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: qdrant-haystack
|
|
3
|
-
Version: 10.
|
|
3
|
+
Version: 10.4.0
|
|
4
4
|
Summary: An integration of Qdrant ANN vector database backend with Haystack
|
|
5
5
|
Project-URL: Source, https://github.com/deepset-ai/haystack-core-integrations
|
|
6
6
|
Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/blob/main/integrations/qdrant/README.md
|
|
@@ -19,7 +19,7 @@ Classifier: Programming Language :: Python :: 3.14
|
|
|
19
19
|
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
20
20
|
Classifier: Programming Language :: Python :: Implementation :: PyPy
|
|
21
21
|
Requires-Python: >=3.10
|
|
22
|
-
Requires-Dist: haystack-ai>=2.
|
|
22
|
+
Requires-Dist: haystack-ai>=2.29.0
|
|
23
23
|
Requires-Dist: qdrant-client>=1.12.0
|
|
24
24
|
Description-Content-Type: text/markdown
|
|
25
25
|
|
|
@@ -9,9 +9,12 @@ import glob
|
|
|
9
9
|
|
|
10
10
|
from haystack import Pipeline
|
|
11
11
|
from haystack.components.converters import MarkdownToDocument
|
|
12
|
-
from haystack.components.embedders import SentenceTransformersDocumentEmbedder, SentenceTransformersTextEmbedder
|
|
13
12
|
from haystack.components.preprocessors import DocumentSplitter
|
|
14
13
|
from haystack.components.writers import DocumentWriter
|
|
14
|
+
from haystack_integrations.components.embedders.sentence_transformers import (
|
|
15
|
+
SentenceTransformersDocumentEmbedder,
|
|
16
|
+
SentenceTransformersTextEmbedder,
|
|
17
|
+
)
|
|
15
18
|
|
|
16
19
|
from haystack_integrations.components.retrievers.qdrant import QdrantEmbeddingRetriever
|
|
17
20
|
from haystack_integrations.document_stores.qdrant import QdrantDocumentStore
|
|
@@ -28,7 +28,7 @@ classifiers = [
|
|
|
28
28
|
"Programming Language :: Python :: Implementation :: PyPy",
|
|
29
29
|
]
|
|
30
30
|
dependencies = [
|
|
31
|
-
"haystack-ai>=2.
|
|
31
|
+
"haystack-ai>=2.29.0",
|
|
32
32
|
"qdrant-client>=1.12.0"
|
|
33
33
|
]
|
|
34
34
|
|
|
@@ -71,7 +71,8 @@ dependencies = [
|
|
|
71
71
|
unit = 'pytest -m "not integration" {args:tests}'
|
|
72
72
|
integration = 'pytest -m "integration" {args:tests}'
|
|
73
73
|
all = 'pytest {args:tests}'
|
|
74
|
-
cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x {args:tests}'
|
|
74
|
+
unit-cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x -m "not integration" {args:tests}'
|
|
75
|
+
integration-cov-append-retry = 'pytest --cov=haystack_integrations --cov-append --reruns 3 --reruns-delay 30 -x -m "integration" {args:tests}'
|
|
75
76
|
|
|
76
77
|
types = """mypy -p haystack_integrations.document_stores.qdrant \
|
|
77
78
|
-p haystack_integrations.components.retrievers.qdrant {args}"""
|
|
@@ -93,6 +94,13 @@ select = [
|
|
|
93
94
|
"ARG",
|
|
94
95
|
"B",
|
|
95
96
|
"C",
|
|
97
|
+
"D102", # Missing docstring in public method
|
|
98
|
+
"D103", # Missing docstring in public function
|
|
99
|
+
"D205", # 1 blank line required between summary line and description
|
|
100
|
+
"D209", # Closing triple quotes go to new line
|
|
101
|
+
"D213", # summary lines must be positioned on the second physical line of the docstring
|
|
102
|
+
"D417", # Missing argument descriptions in the docstring
|
|
103
|
+
"D419", # Docstring is empty
|
|
96
104
|
"DTZ",
|
|
97
105
|
"E",
|
|
98
106
|
"EM",
|
|
@@ -135,6 +143,7 @@ ignore = [
|
|
|
135
143
|
"PLR0912",
|
|
136
144
|
"PLR0913",
|
|
137
145
|
"PLR0915",
|
|
146
|
+
"PLR0917",
|
|
138
147
|
# Ignore assertions
|
|
139
148
|
"S101",
|
|
140
149
|
]
|
|
@@ -144,14 +153,15 @@ ban-relative-imports = "parents"
|
|
|
144
153
|
|
|
145
154
|
[tool.ruff.lint.per-file-ignores]
|
|
146
155
|
# Tests can use magic values, assertions, and relative imports
|
|
147
|
-
"tests/**/*" = ["PLR2004", "S101", "TID252", "ANN"]
|
|
156
|
+
"tests/**/*" = ["D", "PLR2004", "S101", "TID252", "ANN"]
|
|
148
157
|
# examples can contain "print" commands
|
|
149
|
-
"examples/**/*" = ["T201"]
|
|
158
|
+
"examples/**/*" = ["D", "T201"]
|
|
150
159
|
|
|
151
160
|
|
|
152
161
|
[tool.coverage.run]
|
|
153
162
|
source = ["haystack_integrations"]
|
|
154
163
|
branch = true
|
|
164
|
+
relative_files = true
|
|
155
165
|
parallel = false
|
|
156
166
|
|
|
157
167
|
|
|
@@ -163,3 +173,10 @@ omit = [
|
|
|
163
173
|
]
|
|
164
174
|
show_missing = true
|
|
165
175
|
exclude_lines = ["no cov", "if __name__ == .__main__.:", "if TYPE_CHECKING:"]
|
|
176
|
+
|
|
177
|
+
[tool.pytest.ini_options]
|
|
178
|
+
addopts = "--strict-markers"
|
|
179
|
+
markers = [
|
|
180
|
+
"integration: integration tests",
|
|
181
|
+
]
|
|
182
|
+
log_cli = true
|
|
@@ -130,6 +130,18 @@ class QdrantEmbeddingRetriever:
|
|
|
130
130
|
data["init_parameters"]["filter_policy"] = FilterPolicy.from_str(filter_policy)
|
|
131
131
|
return default_from_dict(cls, data)
|
|
132
132
|
|
|
133
|
+
def close(self) -> None:
|
|
134
|
+
"""
|
|
135
|
+
Release the synchronous resources of the underlying Document Store.
|
|
136
|
+
"""
|
|
137
|
+
self._document_store.close()
|
|
138
|
+
|
|
139
|
+
async def close_async(self) -> None:
|
|
140
|
+
"""
|
|
141
|
+
Release the asynchronous resources of the underlying Document Store.
|
|
142
|
+
"""
|
|
143
|
+
await self._document_store.close_async()
|
|
144
|
+
|
|
133
145
|
@component.output_types(documents=list[Document])
|
|
134
146
|
def run(
|
|
135
147
|
self,
|
|
@@ -358,6 +370,18 @@ class QdrantSparseEmbeddingRetriever:
|
|
|
358
370
|
data["init_parameters"]["filter_policy"] = FilterPolicy.from_str(filter_policy)
|
|
359
371
|
return default_from_dict(cls, data)
|
|
360
372
|
|
|
373
|
+
def close(self) -> None:
|
|
374
|
+
"""
|
|
375
|
+
Release the synchronous resources of the underlying Document Store.
|
|
376
|
+
"""
|
|
377
|
+
self._document_store.close()
|
|
378
|
+
|
|
379
|
+
async def close_async(self) -> None:
|
|
380
|
+
"""
|
|
381
|
+
Release the asynchronous resources of the underlying Document Store.
|
|
382
|
+
"""
|
|
383
|
+
await self._document_store.close_async()
|
|
384
|
+
|
|
361
385
|
@component.output_types(documents=list[Document])
|
|
362
386
|
def run(
|
|
363
387
|
self,
|
|
@@ -482,8 +506,9 @@ class QdrantSparseEmbeddingRetriever:
|
|
|
482
506
|
@component
|
|
483
507
|
class QdrantHybridRetriever:
|
|
484
508
|
"""
|
|
485
|
-
A component for retrieving documents from
|
|
486
|
-
|
|
509
|
+
A component for retrieving documents from a QdrantDocumentStore using both dense and sparse vectors.
|
|
510
|
+
|
|
511
|
+
Fuses the results using Reciprocal Rank Fusion.
|
|
487
512
|
|
|
488
513
|
Usage example:
|
|
489
514
|
```python
|
|
@@ -595,6 +620,18 @@ class QdrantHybridRetriever:
|
|
|
595
620
|
data["init_parameters"]["filter_policy"] = FilterPolicy.from_str(filter_policy)
|
|
596
621
|
return default_from_dict(cls, data)
|
|
597
622
|
|
|
623
|
+
def close(self) -> None:
|
|
624
|
+
"""
|
|
625
|
+
Release the synchronous resources of the underlying Document Store.
|
|
626
|
+
"""
|
|
627
|
+
self._document_store.close()
|
|
628
|
+
|
|
629
|
+
async def close_async(self) -> None:
|
|
630
|
+
"""
|
|
631
|
+
Release the asynchronous resources of the underlying Document Store.
|
|
632
|
+
"""
|
|
633
|
+
await self._document_store.close_async()
|
|
634
|
+
|
|
598
635
|
@component.output_types(documents=list[Document])
|
|
599
636
|
def run(
|
|
600
637
|
self,
|
|
@@ -18,6 +18,7 @@ def convert_haystack_documents_to_qdrant_points(
|
|
|
18
18
|
*,
|
|
19
19
|
use_sparse_embeddings: bool,
|
|
20
20
|
) -> list[rest.PointStruct]:
|
|
21
|
+
"""Convert a list of Haystack Document objects to Qdrant PointStruct objects."""
|
|
21
22
|
points = []
|
|
22
23
|
for document in documents:
|
|
23
24
|
payload = document.to_dict(flatten=False)
|
|
@@ -61,6 +62,7 @@ QdrantPoint = rest.ScoredPoint | rest.Record
|
|
|
61
62
|
|
|
62
63
|
|
|
63
64
|
def convert_qdrant_point_to_haystack_document(point: QdrantPoint, use_sparse_embeddings: bool) -> Document:
|
|
65
|
+
"""Convert a Qdrant ScoredPoint or Record to a Haystack Document object."""
|
|
64
66
|
payload = point.payload or {}
|
|
65
67
|
payload["score"] = point.score if hasattr(point, "score") else None
|
|
66
68
|
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import inspect
|
|
2
2
|
from collections.abc import AsyncGenerator, Generator
|
|
3
|
+
from contextlib import suppress
|
|
4
|
+
from dataclasses import replace
|
|
3
5
|
from itertools import islice
|
|
4
6
|
from typing import Any, ClassVar, cast
|
|
5
7
|
|
|
@@ -54,8 +56,9 @@ def get_batches_from_generator(iterable: list, n: int) -> Generator:
|
|
|
54
56
|
|
|
55
57
|
class QdrantDocumentStore:
|
|
56
58
|
"""
|
|
57
|
-
A QdrantDocumentStore implementation that you can use with any Qdrant instance
|
|
58
|
-
|
|
59
|
+
A QdrantDocumentStore implementation that you can use with any Qdrant instance.
|
|
60
|
+
|
|
61
|
+
Supports in-memory, disk-persisted, Docker-based, and Qdrant Cloud Cluster deployments.
|
|
59
62
|
|
|
60
63
|
Usage example by creating an in-memory instance:
|
|
61
64
|
|
|
@@ -299,6 +302,24 @@ class QdrantDocumentStore:
|
|
|
299
302
|
self.payload_fields_to_index,
|
|
300
303
|
)
|
|
301
304
|
|
|
305
|
+
def close(self) -> None:
|
|
306
|
+
"""
|
|
307
|
+
Release the associated synchronous resources.
|
|
308
|
+
"""
|
|
309
|
+
if self._client is not None:
|
|
310
|
+
with suppress(Exception):
|
|
311
|
+
self._client.close()
|
|
312
|
+
self._client = None
|
|
313
|
+
|
|
314
|
+
async def close_async(self) -> None:
|
|
315
|
+
"""
|
|
316
|
+
Release the associated asynchronous resources.
|
|
317
|
+
"""
|
|
318
|
+
if self._async_client is not None:
|
|
319
|
+
with suppress(Exception):
|
|
320
|
+
await self._async_client.close()
|
|
321
|
+
self._async_client = None
|
|
322
|
+
|
|
302
323
|
def count_documents(self) -> int:
|
|
303
324
|
"""
|
|
304
325
|
Returns the number of documents present in the Document Store.
|
|
@@ -376,6 +397,7 @@ class QdrantDocumentStore:
|
|
|
376
397
|
) -> int:
|
|
377
398
|
"""
|
|
378
399
|
Writes documents to Qdrant using the specified policy.
|
|
400
|
+
|
|
379
401
|
The QdrantDocumentStore can handle duplicate documents based on the given policy.
|
|
380
402
|
The available policies are:
|
|
381
403
|
- `FAIL`: The operation will raise an error if any document already exists.
|
|
@@ -429,6 +451,7 @@ class QdrantDocumentStore:
|
|
|
429
451
|
) -> int:
|
|
430
452
|
"""
|
|
431
453
|
Asynchronously writes documents to Qdrant using the specified policy.
|
|
454
|
+
|
|
432
455
|
The QdrantDocumentStore can handle duplicate documents based on the given policy.
|
|
433
456
|
The available policies are:
|
|
434
457
|
- `FAIL`: The operation will raise an error if any document already exists.
|
|
@@ -1223,8 +1246,9 @@ class QdrantDocumentStore:
|
|
|
1223
1246
|
self, filters: dict[str, Any], metadata_fields: list[str]
|
|
1224
1247
|
) -> dict[str, int]:
|
|
1225
1248
|
"""
|
|
1226
|
-
Asynchronously returns the number of unique values for each specified metadata field among documents
|
|
1227
|
-
|
|
1249
|
+
Asynchronously returns the number of unique values for each specified metadata field among documents.
|
|
1250
|
+
|
|
1251
|
+
Only documents that match the filters are considered.
|
|
1228
1252
|
|
|
1229
1253
|
:param filters: The filters to restrict the documents considered.
|
|
1230
1254
|
For filter syntax, see [Haystack metadata filtering](https://docs.haystack.deepset.ai/docs/metadata-filtering)
|
|
@@ -1926,8 +1950,9 @@ class QdrantDocumentStore:
|
|
|
1926
1950
|
group_size: int | None = None,
|
|
1927
1951
|
) -> list[Document]:
|
|
1928
1952
|
"""
|
|
1929
|
-
Asynchronously retrieves documents based on dense and sparse embeddings
|
|
1930
|
-
|
|
1953
|
+
Asynchronously retrieves documents based on dense and sparse embeddings.
|
|
1954
|
+
|
|
1955
|
+
Fuses the results using Reciprocal Rank Fusion.
|
|
1931
1956
|
|
|
1932
1957
|
This method is not part of the public interface of `QdrantDocumentStore` and shouldn't be used directly.
|
|
1933
1958
|
Use the `QdrantHybridRetriever` instead.
|
|
@@ -2292,8 +2317,9 @@ class QdrantDocumentStore:
|
|
|
2292
2317
|
policy: DuplicatePolicy | None = None,
|
|
2293
2318
|
) -> list[Document]:
|
|
2294
2319
|
"""
|
|
2295
|
-
Checks whether any of the passed documents is already existing in the chosen index
|
|
2296
|
-
|
|
2320
|
+
Checks whether any of the passed documents is already existing in the chosen index.
|
|
2321
|
+
|
|
2322
|
+
Returns a list of documents that are not in the index yet.
|
|
2297
2323
|
|
|
2298
2324
|
:param documents: A list of Haystack Document objects.
|
|
2299
2325
|
:param policy: The duplicate policy to use when writing documents.
|
|
@@ -2319,9 +2345,9 @@ class QdrantDocumentStore:
|
|
|
2319
2345
|
policy: DuplicatePolicy | None = None,
|
|
2320
2346
|
) -> list[Document]:
|
|
2321
2347
|
"""
|
|
2322
|
-
Asynchronously checks whether any of the passed documents is already existing
|
|
2323
|
-
|
|
2324
|
-
documents that are not in the index yet.
|
|
2348
|
+
Asynchronously checks whether any of the passed documents is already existing in the chosen index.
|
|
2349
|
+
|
|
2350
|
+
Returns a list of documents that are not in the index yet.
|
|
2325
2351
|
|
|
2326
2352
|
:param documents: A list of Haystack Document objects.
|
|
2327
2353
|
:param policy: The duplicate policy to use when writing documents.
|
|
@@ -2468,15 +2494,17 @@ class QdrantDocumentStore:
|
|
|
2468
2494
|
]
|
|
2469
2495
|
|
|
2470
2496
|
if scale_score:
|
|
2471
|
-
|
|
2472
|
-
|
|
2473
|
-
|
|
2474
|
-
|
|
2475
|
-
|
|
2476
|
-
|
|
2477
|
-
|
|
2478
|
-
|
|
2479
|
-
document
|
|
2497
|
+
documents = [
|
|
2498
|
+
replace(
|
|
2499
|
+
document,
|
|
2500
|
+
score=(document.score + 1) / 2
|
|
2501
|
+
if self.similarity == "cosine"
|
|
2502
|
+
else float(1 / (1 + exp(-document.score / 100))),
|
|
2503
|
+
)
|
|
2504
|
+
if document.score is not None
|
|
2505
|
+
else document
|
|
2506
|
+
for document in documents
|
|
2507
|
+
]
|
|
2480
2508
|
|
|
2481
2509
|
return documents
|
|
2482
2510
|
|
|
@@ -9,7 +9,8 @@ from qdrant_client.http import models
|
|
|
9
9
|
def convert_filters_to_qdrant(
|
|
10
10
|
filter_term: list[dict[str, Any]] | dict[str, Any] | models.Filter | None = None,
|
|
11
11
|
) -> models.Filter | None:
|
|
12
|
-
"""
|
|
12
|
+
"""
|
|
13
|
+
Converts Haystack filters to the format used by Qdrant.
|
|
13
14
|
|
|
14
15
|
:param filter_term: the haystack filter to be converted to qdrant.
|
|
15
16
|
:returns: a single Qdrant Filter or None.
|
|
@@ -228,6 +229,7 @@ def _build_gte_condition(key: str, value: str | float | int) -> models.Condition
|
|
|
228
229
|
|
|
229
230
|
|
|
230
231
|
def is_datetime_string(value: str) -> bool:
|
|
232
|
+
"""Return True if the given string can be parsed as an ISO 8601 datetime, False otherwise."""
|
|
231
233
|
try:
|
|
232
234
|
datetime.fromisoformat(value)
|
|
233
235
|
return True
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
from types import SimpleNamespace
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pytest
|
|
5
|
+
from haystack.dataclasses import Document, SparseEmbedding
|
|
6
|
+
from qdrant_client.http import models as rest
|
|
7
|
+
|
|
8
|
+
from haystack_integrations.document_stores.qdrant.converters import (
|
|
9
|
+
DENSE_VECTORS_NAME,
|
|
10
|
+
SPARSE_VECTORS_NAME,
|
|
11
|
+
convert_haystack_documents_to_qdrant_points,
|
|
12
|
+
convert_id,
|
|
13
|
+
convert_qdrant_point_to_haystack_document,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def test_convert_id_is_deterministic():
|
|
18
|
+
first_id = convert_id("test-id")
|
|
19
|
+
second_id = convert_id("test-id")
|
|
20
|
+
assert first_id == second_id
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_point_to_document_reverts_proper_structure_from_record_with_sparse():
|
|
24
|
+
point = rest.Record(
|
|
25
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
26
|
+
payload={
|
|
27
|
+
"id": "my-id",
|
|
28
|
+
"id_hash_keys": ["content"],
|
|
29
|
+
"content": "Lorem ipsum",
|
|
30
|
+
"content_type": "text",
|
|
31
|
+
"meta": {
|
|
32
|
+
"test_field": 1,
|
|
33
|
+
},
|
|
34
|
+
},
|
|
35
|
+
vector={
|
|
36
|
+
"text-dense": [1.0, 0.0, 0.0, 0.0],
|
|
37
|
+
"text-sparse": {"indices": [7, 1024, 367], "values": [0.1, 0.98, 0.33]},
|
|
38
|
+
},
|
|
39
|
+
)
|
|
40
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
|
|
41
|
+
assert "my-id" == document.id
|
|
42
|
+
assert "Lorem ipsum" == document.content
|
|
43
|
+
assert "text" == document.content_type
|
|
44
|
+
assert {"indices": [7, 1024, 367], "values": [0.1, 0.98, 0.33]} == document.sparse_embedding.to_dict()
|
|
45
|
+
assert {"test_field": 1} == document.meta
|
|
46
|
+
assert 0.0 == np.sum(np.array([1.0, 0.0, 0.0, 0.0]) - document.embedding)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def test_point_to_document_reverts_proper_structure_from_record_without_sparse():
|
|
50
|
+
point = rest.Record(
|
|
51
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
52
|
+
payload={
|
|
53
|
+
"id": "my-id",
|
|
54
|
+
"id_hash_keys": ["content"],
|
|
55
|
+
"content": "Lorem ipsum",
|
|
56
|
+
"content_type": "text",
|
|
57
|
+
"meta": {
|
|
58
|
+
"test_field": 1,
|
|
59
|
+
},
|
|
60
|
+
},
|
|
61
|
+
vector=[1.0, 0.0, 0.0, 0.0],
|
|
62
|
+
)
|
|
63
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=False)
|
|
64
|
+
assert "my-id" == document.id
|
|
65
|
+
assert "Lorem ipsum" == document.content
|
|
66
|
+
assert "text" == document.content_type
|
|
67
|
+
assert document.sparse_embedding is None
|
|
68
|
+
assert {"test_field": 1} == document.meta
|
|
69
|
+
assert 0.0 == np.sum(np.array([1.0, 0.0, 0.0, 0.0]) - document.embedding)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def test_point_to_document_with_sparse_enabled_but_vector_none():
|
|
73
|
+
point = rest.Record(
|
|
74
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
75
|
+
payload={"id": "my-id", "content": "Lorem"},
|
|
76
|
+
vector=None,
|
|
77
|
+
)
|
|
78
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
|
|
79
|
+
assert document.embedding is None
|
|
80
|
+
assert document.sparse_embedding is None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_point_to_document_preserves_score_from_scored_point():
|
|
84
|
+
point = rest.ScoredPoint(
|
|
85
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
86
|
+
payload={"id": "my-id", "content": "Lorem"},
|
|
87
|
+
vector=[0.1, 0.2],
|
|
88
|
+
score=0.75,
|
|
89
|
+
version=0,
|
|
90
|
+
)
|
|
91
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=False)
|
|
92
|
+
assert document.score == 0.75
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_convert_haystack_documents_to_qdrant_points_without_sparse():
|
|
96
|
+
doc = Document(content="hello", embedding=[0.1, 0.2, 0.3])
|
|
97
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=False)
|
|
98
|
+
assert len(points) == 1
|
|
99
|
+
assert points[0].vector == [0.1, 0.2, 0.3]
|
|
100
|
+
assert points[0].payload["content"] == "hello"
|
|
101
|
+
assert "embedding" not in points[0].payload
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_convert_haystack_documents_to_qdrant_points_without_sparse_without_embedding():
|
|
105
|
+
doc = Document(content="hello")
|
|
106
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=False)
|
|
107
|
+
assert points[0].vector == {}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def test_convert_haystack_documents_to_qdrant_points_with_sparse():
|
|
111
|
+
sparse = SparseEmbedding(indices=[0, 5], values=[0.1, 0.7])
|
|
112
|
+
doc = Document(content="hello", embedding=[0.1, 0.2], sparse_embedding=sparse)
|
|
113
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
|
|
114
|
+
assert points[0].vector[DENSE_VECTORS_NAME] == [0.1, 0.2]
|
|
115
|
+
assert isinstance(points[0].vector[SPARSE_VECTORS_NAME], rest.SparseVector)
|
|
116
|
+
assert points[0].vector[SPARSE_VECTORS_NAME].indices == [0, 5]
|
|
117
|
+
assert points[0].vector[SPARSE_VECTORS_NAME].values == [0.1, 0.7]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def test_convert_haystack_documents_to_qdrant_points_with_sparse_only_dense():
|
|
121
|
+
doc = Document(content="hello", embedding=[0.1, 0.2])
|
|
122
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
|
|
123
|
+
assert points[0].vector == {DENSE_VECTORS_NAME: [0.1, 0.2]}
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def test_convert_haystack_documents_to_qdrant_points_with_sparse_no_vectors():
|
|
127
|
+
doc = Document(content="hello")
|
|
128
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
|
|
129
|
+
assert points[0].vector == {}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
@pytest.mark.parametrize(
|
|
133
|
+
"vector",
|
|
134
|
+
[
|
|
135
|
+
{DENSE_VECTORS_NAME: [0.1, 0.2]},
|
|
136
|
+
{DENSE_VECTORS_NAME: [0.1, 0.2], SPARSE_VECTORS_NAME: {"indices": [0], "values": [0.5]}},
|
|
137
|
+
],
|
|
138
|
+
ids=["no_sparse_key", "sparse_value_not_sparse_vector_instance"],
|
|
139
|
+
)
|
|
140
|
+
def test_point_to_document_sparse_vector_edge_cases(vector):
|
|
141
|
+
point = SimpleNamespace(id="x", payload={"id": "x", "content": "x"}, vector=vector)
|
|
142
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
|
|
143
|
+
assert document.embedding == [0.1, 0.2]
|
|
144
|
+
assert document.sparse_embedding is None
|