qdrant-haystack 10.3.0__tar.gz → 10.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/CHANGELOG.md +20 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/PKG-INFO +2 -2
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/examples/embedding_retrieval.py +4 -1
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/pyproject.toml +20 -4
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/components/retrievers/qdrant/retriever.py +3 -2
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/converters.py +2 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/document_store.py +29 -20
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/filters.py +3 -1
- qdrant_haystack-10.3.1/tests/test_converters.py +144 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_document_store.py +323 -130
- qdrant_haystack-10.3.1/tests/test_document_store_async.py +311 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_embedding_retriever.py +101 -5
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_filters.py +80 -1
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_hybrid_retriever.py +26 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_sparse_embedding_retriever.py +74 -11
- qdrant_haystack-10.3.0/tests/test_converters.py +0 -62
- qdrant_haystack-10.3.0/tests/test_document_store_async.py +0 -640
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/.gitignore +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/LICENSE.txt +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/README.md +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/pydoc/config_docusaurus.yml +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/components/retrievers/py.typed +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/components/retrievers/qdrant/__init__.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/py.typed +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/__init__.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/migrate_to_sparse.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/__init__.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/conftest.py +0 -0
- {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_dict_converters.py +0 -0
|
@@ -1,5 +1,25 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [integrations/qdrant-v10.3.0] - 2026-03-23
|
|
4
|
+
|
|
5
|
+
### 📚 Documentation
|
|
6
|
+
|
|
7
|
+
- Simplify pydoc configs (#2855)
|
|
8
|
+
|
|
9
|
+
### 🧪 Testing
|
|
10
|
+
|
|
11
|
+
- Replacing each `DocumentStore` specific tests and used the generalised ones from `haystack.testing.document_store` (#2812)
|
|
12
|
+
- Test compatible integrations with python 3.14; update pyproject (#3001)
|
|
13
|
+
|
|
14
|
+
### 🧹 Chores
|
|
15
|
+
|
|
16
|
+
- Standardize author mentions (#2897)
|
|
17
|
+
- Add ANN ruff ruleset to optimum, paddleocr, pgvector, pinecone, pyversity, qdrant, ragas, snowflake (#2992)
|
|
18
|
+
|
|
19
|
+
### 🌀 Miscellaneous
|
|
20
|
+
|
|
21
|
+
- !test: `QdrantDocumentStore` use Mixin tests + updated signature `get_metadata_fields_info(self) -> dict[str, dict[str, str]]`: (#3004)
|
|
22
|
+
|
|
3
23
|
## [integrations/qdrant-v10.2.1] - 2026-02-02
|
|
4
24
|
|
|
5
25
|
### 📚 Documentation
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: qdrant-haystack
|
|
3
|
-
Version: 10.3.
|
|
3
|
+
Version: 10.3.1
|
|
4
4
|
Summary: An integration of Qdrant ANN vector database backend with Haystack
|
|
5
5
|
Project-URL: Source, https://github.com/deepset-ai/haystack-core-integrations
|
|
6
6
|
Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/blob/main/integrations/qdrant/README.md
|
|
@@ -19,7 +19,7 @@ Classifier: Programming Language :: Python :: 3.14
|
|
|
19
19
|
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
20
20
|
Classifier: Programming Language :: Python :: Implementation :: PyPy
|
|
21
21
|
Requires-Python: >=3.10
|
|
22
|
-
Requires-Dist: haystack-ai>=2.
|
|
22
|
+
Requires-Dist: haystack-ai>=2.29.0
|
|
23
23
|
Requires-Dist: qdrant-client>=1.12.0
|
|
24
24
|
Description-Content-Type: text/markdown
|
|
25
25
|
|
|
@@ -9,9 +9,12 @@ import glob
|
|
|
9
9
|
|
|
10
10
|
from haystack import Pipeline
|
|
11
11
|
from haystack.components.converters import MarkdownToDocument
|
|
12
|
-
from haystack.components.embedders import SentenceTransformersDocumentEmbedder, SentenceTransformersTextEmbedder
|
|
13
12
|
from haystack.components.preprocessors import DocumentSplitter
|
|
14
13
|
from haystack.components.writers import DocumentWriter
|
|
14
|
+
from haystack_integrations.components.embedders.sentence_transformers import (
|
|
15
|
+
SentenceTransformersDocumentEmbedder,
|
|
16
|
+
SentenceTransformersTextEmbedder,
|
|
17
|
+
)
|
|
15
18
|
|
|
16
19
|
from haystack_integrations.components.retrievers.qdrant import QdrantEmbeddingRetriever
|
|
17
20
|
from haystack_integrations.document_stores.qdrant import QdrantDocumentStore
|
|
@@ -28,7 +28,7 @@ classifiers = [
|
|
|
28
28
|
"Programming Language :: Python :: Implementation :: PyPy",
|
|
29
29
|
]
|
|
30
30
|
dependencies = [
|
|
31
|
-
"haystack-ai>=2.
|
|
31
|
+
"haystack-ai>=2.29.0",
|
|
32
32
|
"qdrant-client>=1.12.0"
|
|
33
33
|
]
|
|
34
34
|
|
|
@@ -71,7 +71,8 @@ dependencies = [
|
|
|
71
71
|
unit = 'pytest -m "not integration" {args:tests}'
|
|
72
72
|
integration = 'pytest -m "integration" {args:tests}'
|
|
73
73
|
all = 'pytest {args:tests}'
|
|
74
|
-
cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x {args:tests}'
|
|
74
|
+
unit-cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x -m "not integration" {args:tests}'
|
|
75
|
+
integration-cov-append-retry = 'pytest --cov=haystack_integrations --cov-append --reruns 3 --reruns-delay 30 -x -m "integration" {args:tests}'
|
|
75
76
|
|
|
76
77
|
types = """mypy -p haystack_integrations.document_stores.qdrant \
|
|
77
78
|
-p haystack_integrations.components.retrievers.qdrant {args}"""
|
|
@@ -93,6 +94,13 @@ select = [
|
|
|
93
94
|
"ARG",
|
|
94
95
|
"B",
|
|
95
96
|
"C",
|
|
97
|
+
"D102", # Missing docstring in public method
|
|
98
|
+
"D103", # Missing docstring in public function
|
|
99
|
+
"D205", # 1 blank line required between summary line and description
|
|
100
|
+
"D209", # Closing triple quotes go to new line
|
|
101
|
+
"D213", # summary lines must be positioned on the second physical line of the docstring
|
|
102
|
+
"D417", # Missing argument descriptions in the docstring
|
|
103
|
+
"D419", # Docstring is empty
|
|
96
104
|
"DTZ",
|
|
97
105
|
"E",
|
|
98
106
|
"EM",
|
|
@@ -144,14 +152,15 @@ ban-relative-imports = "parents"
|
|
|
144
152
|
|
|
145
153
|
[tool.ruff.lint.per-file-ignores]
|
|
146
154
|
# Tests can use magic values, assertions, and relative imports
|
|
147
|
-
"tests/**/*" = ["PLR2004", "S101", "TID252", "ANN"]
|
|
155
|
+
"tests/**/*" = ["D", "PLR2004", "S101", "TID252", "ANN"]
|
|
148
156
|
# examples can contain "print" commands
|
|
149
|
-
"examples/**/*" = ["T201"]
|
|
157
|
+
"examples/**/*" = ["D", "T201"]
|
|
150
158
|
|
|
151
159
|
|
|
152
160
|
[tool.coverage.run]
|
|
153
161
|
source = ["haystack_integrations"]
|
|
154
162
|
branch = true
|
|
163
|
+
relative_files = true
|
|
155
164
|
parallel = false
|
|
156
165
|
|
|
157
166
|
|
|
@@ -163,3 +172,10 @@ omit = [
|
|
|
163
172
|
]
|
|
164
173
|
show_missing = true
|
|
165
174
|
exclude_lines = ["no cov", "if __name__ == .__main__.:", "if TYPE_CHECKING:"]
|
|
175
|
+
|
|
176
|
+
[tool.pytest.ini_options]
|
|
177
|
+
addopts = "--strict-markers"
|
|
178
|
+
markers = [
|
|
179
|
+
"integration: integration tests",
|
|
180
|
+
]
|
|
181
|
+
log_cli = true
|
|
@@ -482,8 +482,9 @@ class QdrantSparseEmbeddingRetriever:
|
|
|
482
482
|
@component
|
|
483
483
|
class QdrantHybridRetriever:
|
|
484
484
|
"""
|
|
485
|
-
A component for retrieving documents from
|
|
486
|
-
|
|
485
|
+
A component for retrieving documents from a QdrantDocumentStore using both dense and sparse vectors.
|
|
486
|
+
|
|
487
|
+
Fuses the results using Reciprocal Rank Fusion.
|
|
487
488
|
|
|
488
489
|
Usage example:
|
|
489
490
|
```python
|
|
@@ -18,6 +18,7 @@ def convert_haystack_documents_to_qdrant_points(
|
|
|
18
18
|
*,
|
|
19
19
|
use_sparse_embeddings: bool,
|
|
20
20
|
) -> list[rest.PointStruct]:
|
|
21
|
+
"""Convert a list of Haystack Document objects to Qdrant PointStruct objects."""
|
|
21
22
|
points = []
|
|
22
23
|
for document in documents:
|
|
23
24
|
payload = document.to_dict(flatten=False)
|
|
@@ -61,6 +62,7 @@ QdrantPoint = rest.ScoredPoint | rest.Record
|
|
|
61
62
|
|
|
62
63
|
|
|
63
64
|
def convert_qdrant_point_to_haystack_document(point: QdrantPoint, use_sparse_embeddings: bool) -> Document:
|
|
65
|
+
"""Convert a Qdrant ScoredPoint or Record to a Haystack Document object."""
|
|
64
66
|
payload = point.payload or {}
|
|
65
67
|
payload["score"] = point.score if hasattr(point, "score") else None
|
|
66
68
|
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import inspect
|
|
2
2
|
from collections.abc import AsyncGenerator, Generator
|
|
3
|
+
from dataclasses import replace
|
|
3
4
|
from itertools import islice
|
|
4
5
|
from typing import Any, ClassVar, cast
|
|
5
6
|
|
|
@@ -54,8 +55,9 @@ def get_batches_from_generator(iterable: list, n: int) -> Generator:
|
|
|
54
55
|
|
|
55
56
|
class QdrantDocumentStore:
|
|
56
57
|
"""
|
|
57
|
-
A QdrantDocumentStore implementation that you can use with any Qdrant instance
|
|
58
|
-
|
|
58
|
+
A QdrantDocumentStore implementation that you can use with any Qdrant instance.
|
|
59
|
+
|
|
60
|
+
Supports in-memory, disk-persisted, Docker-based, and Qdrant Cloud Cluster deployments.
|
|
59
61
|
|
|
60
62
|
Usage example by creating an in-memory instance:
|
|
61
63
|
|
|
@@ -376,6 +378,7 @@ class QdrantDocumentStore:
|
|
|
376
378
|
) -> int:
|
|
377
379
|
"""
|
|
378
380
|
Writes documents to Qdrant using the specified policy.
|
|
381
|
+
|
|
379
382
|
The QdrantDocumentStore can handle duplicate documents based on the given policy.
|
|
380
383
|
The available policies are:
|
|
381
384
|
- `FAIL`: The operation will raise an error if any document already exists.
|
|
@@ -429,6 +432,7 @@ class QdrantDocumentStore:
|
|
|
429
432
|
) -> int:
|
|
430
433
|
"""
|
|
431
434
|
Asynchronously writes documents to Qdrant using the specified policy.
|
|
435
|
+
|
|
432
436
|
The QdrantDocumentStore can handle duplicate documents based on the given policy.
|
|
433
437
|
The available policies are:
|
|
434
438
|
- `FAIL`: The operation will raise an error if any document already exists.
|
|
@@ -1223,8 +1227,9 @@ class QdrantDocumentStore:
|
|
|
1223
1227
|
self, filters: dict[str, Any], metadata_fields: list[str]
|
|
1224
1228
|
) -> dict[str, int]:
|
|
1225
1229
|
"""
|
|
1226
|
-
Asynchronously returns the number of unique values for each specified metadata field among documents
|
|
1227
|
-
|
|
1230
|
+
Asynchronously returns the number of unique values for each specified metadata field among documents.
|
|
1231
|
+
|
|
1232
|
+
Only documents that match the filters are considered.
|
|
1228
1233
|
|
|
1229
1234
|
:param filters: The filters to restrict the documents considered.
|
|
1230
1235
|
For filter syntax, see [Haystack metadata filtering](https://docs.haystack.deepset.ai/docs/metadata-filtering)
|
|
@@ -1926,8 +1931,9 @@ class QdrantDocumentStore:
|
|
|
1926
1931
|
group_size: int | None = None,
|
|
1927
1932
|
) -> list[Document]:
|
|
1928
1933
|
"""
|
|
1929
|
-
Asynchronously retrieves documents based on dense and sparse embeddings
|
|
1930
|
-
|
|
1934
|
+
Asynchronously retrieves documents based on dense and sparse embeddings.
|
|
1935
|
+
|
|
1936
|
+
Fuses the results using Reciprocal Rank Fusion.
|
|
1931
1937
|
|
|
1932
1938
|
This method is not part of the public interface of `QdrantDocumentStore` and shouldn't be used directly.
|
|
1933
1939
|
Use the `QdrantHybridRetriever` instead.
|
|
@@ -2292,8 +2298,9 @@ class QdrantDocumentStore:
|
|
|
2292
2298
|
policy: DuplicatePolicy | None = None,
|
|
2293
2299
|
) -> list[Document]:
|
|
2294
2300
|
"""
|
|
2295
|
-
Checks whether any of the passed documents is already existing in the chosen index
|
|
2296
|
-
|
|
2301
|
+
Checks whether any of the passed documents is already existing in the chosen index.
|
|
2302
|
+
|
|
2303
|
+
Returns a list of documents that are not in the index yet.
|
|
2297
2304
|
|
|
2298
2305
|
:param documents: A list of Haystack Document objects.
|
|
2299
2306
|
:param policy: The duplicate policy to use when writing documents.
|
|
@@ -2319,9 +2326,9 @@ class QdrantDocumentStore:
|
|
|
2319
2326
|
policy: DuplicatePolicy | None = None,
|
|
2320
2327
|
) -> list[Document]:
|
|
2321
2328
|
"""
|
|
2322
|
-
Asynchronously checks whether any of the passed documents is already existing
|
|
2323
|
-
|
|
2324
|
-
documents that are not in the index yet.
|
|
2329
|
+
Asynchronously checks whether any of the passed documents is already existing in the chosen index.
|
|
2330
|
+
|
|
2331
|
+
Returns a list of documents that are not in the index yet.
|
|
2325
2332
|
|
|
2326
2333
|
:param documents: A list of Haystack Document objects.
|
|
2327
2334
|
:param policy: The duplicate policy to use when writing documents.
|
|
@@ -2468,15 +2475,17 @@ class QdrantDocumentStore:
|
|
|
2468
2475
|
]
|
|
2469
2476
|
|
|
2470
2477
|
if scale_score:
|
|
2471
|
-
|
|
2472
|
-
|
|
2473
|
-
|
|
2474
|
-
|
|
2475
|
-
|
|
2476
|
-
|
|
2477
|
-
|
|
2478
|
-
|
|
2479
|
-
document
|
|
2478
|
+
documents = [
|
|
2479
|
+
replace(
|
|
2480
|
+
document,
|
|
2481
|
+
score=(document.score + 1) / 2
|
|
2482
|
+
if self.similarity == "cosine"
|
|
2483
|
+
else float(1 / (1 + exp(-document.score / 100))),
|
|
2484
|
+
)
|
|
2485
|
+
if document.score is not None
|
|
2486
|
+
else document
|
|
2487
|
+
for document in documents
|
|
2488
|
+
]
|
|
2480
2489
|
|
|
2481
2490
|
return documents
|
|
2482
2491
|
|
|
@@ -9,7 +9,8 @@ from qdrant_client.http import models
|
|
|
9
9
|
def convert_filters_to_qdrant(
|
|
10
10
|
filter_term: list[dict[str, Any]] | dict[str, Any] | models.Filter | None = None,
|
|
11
11
|
) -> models.Filter | None:
|
|
12
|
-
"""
|
|
12
|
+
"""
|
|
13
|
+
Converts Haystack filters to the format used by Qdrant.
|
|
13
14
|
|
|
14
15
|
:param filter_term: the haystack filter to be converted to qdrant.
|
|
15
16
|
:returns: a single Qdrant Filter or None.
|
|
@@ -228,6 +229,7 @@ def _build_gte_condition(key: str, value: str | float | int) -> models.Condition
|
|
|
228
229
|
|
|
229
230
|
|
|
230
231
|
def is_datetime_string(value: str) -> bool:
|
|
232
|
+
"""Return True if the given string can be parsed as an ISO 8601 datetime, False otherwise."""
|
|
231
233
|
try:
|
|
232
234
|
datetime.fromisoformat(value)
|
|
233
235
|
return True
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
from types import SimpleNamespace
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pytest
|
|
5
|
+
from haystack.dataclasses import Document, SparseEmbedding
|
|
6
|
+
from qdrant_client.http import models as rest
|
|
7
|
+
|
|
8
|
+
from haystack_integrations.document_stores.qdrant.converters import (
|
|
9
|
+
DENSE_VECTORS_NAME,
|
|
10
|
+
SPARSE_VECTORS_NAME,
|
|
11
|
+
convert_haystack_documents_to_qdrant_points,
|
|
12
|
+
convert_id,
|
|
13
|
+
convert_qdrant_point_to_haystack_document,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def test_convert_id_is_deterministic():
|
|
18
|
+
first_id = convert_id("test-id")
|
|
19
|
+
second_id = convert_id("test-id")
|
|
20
|
+
assert first_id == second_id
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_point_to_document_reverts_proper_structure_from_record_with_sparse():
|
|
24
|
+
point = rest.Record(
|
|
25
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
26
|
+
payload={
|
|
27
|
+
"id": "my-id",
|
|
28
|
+
"id_hash_keys": ["content"],
|
|
29
|
+
"content": "Lorem ipsum",
|
|
30
|
+
"content_type": "text",
|
|
31
|
+
"meta": {
|
|
32
|
+
"test_field": 1,
|
|
33
|
+
},
|
|
34
|
+
},
|
|
35
|
+
vector={
|
|
36
|
+
"text-dense": [1.0, 0.0, 0.0, 0.0],
|
|
37
|
+
"text-sparse": {"indices": [7, 1024, 367], "values": [0.1, 0.98, 0.33]},
|
|
38
|
+
},
|
|
39
|
+
)
|
|
40
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
|
|
41
|
+
assert "my-id" == document.id
|
|
42
|
+
assert "Lorem ipsum" == document.content
|
|
43
|
+
assert "text" == document.content_type
|
|
44
|
+
assert {"indices": [7, 1024, 367], "values": [0.1, 0.98, 0.33]} == document.sparse_embedding.to_dict()
|
|
45
|
+
assert {"test_field": 1} == document.meta
|
|
46
|
+
assert 0.0 == np.sum(np.array([1.0, 0.0, 0.0, 0.0]) - document.embedding)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def test_point_to_document_reverts_proper_structure_from_record_without_sparse():
|
|
50
|
+
point = rest.Record(
|
|
51
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
52
|
+
payload={
|
|
53
|
+
"id": "my-id",
|
|
54
|
+
"id_hash_keys": ["content"],
|
|
55
|
+
"content": "Lorem ipsum",
|
|
56
|
+
"content_type": "text",
|
|
57
|
+
"meta": {
|
|
58
|
+
"test_field": 1,
|
|
59
|
+
},
|
|
60
|
+
},
|
|
61
|
+
vector=[1.0, 0.0, 0.0, 0.0],
|
|
62
|
+
)
|
|
63
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=False)
|
|
64
|
+
assert "my-id" == document.id
|
|
65
|
+
assert "Lorem ipsum" == document.content
|
|
66
|
+
assert "text" == document.content_type
|
|
67
|
+
assert document.sparse_embedding is None
|
|
68
|
+
assert {"test_field": 1} == document.meta
|
|
69
|
+
assert 0.0 == np.sum(np.array([1.0, 0.0, 0.0, 0.0]) - document.embedding)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def test_point_to_document_with_sparse_enabled_but_vector_none():
|
|
73
|
+
point = rest.Record(
|
|
74
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
75
|
+
payload={"id": "my-id", "content": "Lorem"},
|
|
76
|
+
vector=None,
|
|
77
|
+
)
|
|
78
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
|
|
79
|
+
assert document.embedding is None
|
|
80
|
+
assert document.sparse_embedding is None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_point_to_document_preserves_score_from_scored_point():
|
|
84
|
+
point = rest.ScoredPoint(
|
|
85
|
+
id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
|
|
86
|
+
payload={"id": "my-id", "content": "Lorem"},
|
|
87
|
+
vector=[0.1, 0.2],
|
|
88
|
+
score=0.75,
|
|
89
|
+
version=0,
|
|
90
|
+
)
|
|
91
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=False)
|
|
92
|
+
assert document.score == 0.75
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_convert_haystack_documents_to_qdrant_points_without_sparse():
|
|
96
|
+
doc = Document(content="hello", embedding=[0.1, 0.2, 0.3])
|
|
97
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=False)
|
|
98
|
+
assert len(points) == 1
|
|
99
|
+
assert points[0].vector == [0.1, 0.2, 0.3]
|
|
100
|
+
assert points[0].payload["content"] == "hello"
|
|
101
|
+
assert "embedding" not in points[0].payload
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_convert_haystack_documents_to_qdrant_points_without_sparse_without_embedding():
|
|
105
|
+
doc = Document(content="hello")
|
|
106
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=False)
|
|
107
|
+
assert points[0].vector == {}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def test_convert_haystack_documents_to_qdrant_points_with_sparse():
|
|
111
|
+
sparse = SparseEmbedding(indices=[0, 5], values=[0.1, 0.7])
|
|
112
|
+
doc = Document(content="hello", embedding=[0.1, 0.2], sparse_embedding=sparse)
|
|
113
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
|
|
114
|
+
assert points[0].vector[DENSE_VECTORS_NAME] == [0.1, 0.2]
|
|
115
|
+
assert isinstance(points[0].vector[SPARSE_VECTORS_NAME], rest.SparseVector)
|
|
116
|
+
assert points[0].vector[SPARSE_VECTORS_NAME].indices == [0, 5]
|
|
117
|
+
assert points[0].vector[SPARSE_VECTORS_NAME].values == [0.1, 0.7]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def test_convert_haystack_documents_to_qdrant_points_with_sparse_only_dense():
|
|
121
|
+
doc = Document(content="hello", embedding=[0.1, 0.2])
|
|
122
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
|
|
123
|
+
assert points[0].vector == {DENSE_VECTORS_NAME: [0.1, 0.2]}
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def test_convert_haystack_documents_to_qdrant_points_with_sparse_no_vectors():
|
|
127
|
+
doc = Document(content="hello")
|
|
128
|
+
points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
|
|
129
|
+
assert points[0].vector == {}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
@pytest.mark.parametrize(
|
|
133
|
+
"vector",
|
|
134
|
+
[
|
|
135
|
+
{DENSE_VECTORS_NAME: [0.1, 0.2]},
|
|
136
|
+
{DENSE_VECTORS_NAME: [0.1, 0.2], SPARSE_VECTORS_NAME: {"indices": [0], "values": [0.5]}},
|
|
137
|
+
],
|
|
138
|
+
ids=["no_sparse_key", "sparse_value_not_sparse_vector_instance"],
|
|
139
|
+
)
|
|
140
|
+
def test_point_to_document_sparse_vector_edge_cases(vector):
|
|
141
|
+
point = SimpleNamespace(id="x", payload={"id": "x", "content": "x"}, vector=vector)
|
|
142
|
+
document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
|
|
143
|
+
assert document.embedding == [0.1, 0.2]
|
|
144
|
+
assert document.sparse_embedding is None
|