qdrant-haystack 10.3.0__tar.gz → 10.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/CHANGELOG.md +20 -0
  2. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/PKG-INFO +2 -2
  3. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/examples/embedding_retrieval.py +4 -1
  4. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/pyproject.toml +20 -4
  5. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/components/retrievers/qdrant/retriever.py +3 -2
  6. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/converters.py +2 -0
  7. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/document_store.py +29 -20
  8. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/filters.py +3 -1
  9. qdrant_haystack-10.3.1/tests/test_converters.py +144 -0
  10. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_document_store.py +323 -130
  11. qdrant_haystack-10.3.1/tests/test_document_store_async.py +311 -0
  12. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_embedding_retriever.py +101 -5
  13. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_filters.py +80 -1
  14. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_hybrid_retriever.py +26 -0
  15. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_sparse_embedding_retriever.py +74 -11
  16. qdrant_haystack-10.3.0/tests/test_converters.py +0 -62
  17. qdrant_haystack-10.3.0/tests/test_document_store_async.py +0 -640
  18. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/.gitignore +0 -0
  19. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/LICENSE.txt +0 -0
  20. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/README.md +0 -0
  21. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/pydoc/config_docusaurus.yml +0 -0
  22. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/components/retrievers/py.typed +0 -0
  23. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/components/retrievers/qdrant/__init__.py +0 -0
  24. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/py.typed +0 -0
  25. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/__init__.py +0 -0
  26. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/src/haystack_integrations/document_stores/qdrant/migrate_to_sparse.py +0 -0
  27. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/__init__.py +0 -0
  28. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/conftest.py +0 -0
  29. {qdrant_haystack-10.3.0 → qdrant_haystack-10.3.1}/tests/test_dict_converters.py +0 -0
@@ -1,5 +1,25 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/qdrant-v10.3.0] - 2026-03-23
4
+
5
+ ### 📚 Documentation
6
+
7
+ - Simplify pydoc configs (#2855)
8
+
9
+ ### 🧪 Testing
10
+
11
+ - Replacing each `DocumentStore` specific tests and used the generalised ones from `haystack.testing.document_store` (#2812)
12
+ - Test compatible integrations with python 3.14; update pyproject (#3001)
13
+
14
+ ### 🧹 Chores
15
+
16
+ - Standardize author mentions (#2897)
17
+ - Add ANN ruff ruleset to optimum, paddleocr, pgvector, pinecone, pyversity, qdrant, ragas, snowflake (#2992)
18
+
19
+ ### 🌀 Miscellaneous
20
+
21
+ - !test: `QdrantDocumentStore` use Mixin tests + updated signature `get_metadata_fields_info(self) -> dict[str, dict[str, str]]`: (#3004)
22
+
3
23
  ## [integrations/qdrant-v10.2.1] - 2026-02-02
4
24
 
5
25
  ### 📚 Documentation
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: qdrant-haystack
3
- Version: 10.3.0
3
+ Version: 10.3.1
4
4
  Summary: An integration of Qdrant ANN vector database backend with Haystack
5
5
  Project-URL: Source, https://github.com/deepset-ai/haystack-core-integrations
6
6
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/blob/main/integrations/qdrant/README.md
@@ -19,7 +19,7 @@ Classifier: Programming Language :: Python :: 3.14
19
19
  Classifier: Programming Language :: Python :: Implementation :: CPython
20
20
  Classifier: Programming Language :: Python :: Implementation :: PyPy
21
21
  Requires-Python: >=3.10
22
- Requires-Dist: haystack-ai>=2.26.1
22
+ Requires-Dist: haystack-ai>=2.29.0
23
23
  Requires-Dist: qdrant-client>=1.12.0
24
24
  Description-Content-Type: text/markdown
25
25
 
@@ -9,9 +9,12 @@ import glob
9
9
 
10
10
  from haystack import Pipeline
11
11
  from haystack.components.converters import MarkdownToDocument
12
- from haystack.components.embedders import SentenceTransformersDocumentEmbedder, SentenceTransformersTextEmbedder
13
12
  from haystack.components.preprocessors import DocumentSplitter
14
13
  from haystack.components.writers import DocumentWriter
14
+ from haystack_integrations.components.embedders.sentence_transformers import (
15
+ SentenceTransformersDocumentEmbedder,
16
+ SentenceTransformersTextEmbedder,
17
+ )
15
18
 
16
19
  from haystack_integrations.components.retrievers.qdrant import QdrantEmbeddingRetriever
17
20
  from haystack_integrations.document_stores.qdrant import QdrantDocumentStore
@@ -28,7 +28,7 @@ classifiers = [
28
28
  "Programming Language :: Python :: Implementation :: PyPy",
29
29
  ]
30
30
  dependencies = [
31
- "haystack-ai>=2.26.1",
31
+ "haystack-ai>=2.29.0",
32
32
  "qdrant-client>=1.12.0"
33
33
  ]
34
34
 
@@ -71,7 +71,8 @@ dependencies = [
71
71
  unit = 'pytest -m "not integration" {args:tests}'
72
72
  integration = 'pytest -m "integration" {args:tests}'
73
73
  all = 'pytest {args:tests}'
74
- cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x {args:tests}'
74
+ unit-cov-retry = 'pytest --cov=haystack_integrations --reruns 3 --reruns-delay 30 -x -m "not integration" {args:tests}'
75
+ integration-cov-append-retry = 'pytest --cov=haystack_integrations --cov-append --reruns 3 --reruns-delay 30 -x -m "integration" {args:tests}'
75
76
 
76
77
  types = """mypy -p haystack_integrations.document_stores.qdrant \
77
78
  -p haystack_integrations.components.retrievers.qdrant {args}"""
@@ -93,6 +94,13 @@ select = [
93
94
  "ARG",
94
95
  "B",
95
96
  "C",
97
+ "D102", # Missing docstring in public method
98
+ "D103", # Missing docstring in public function
99
+ "D205", # 1 blank line required between summary line and description
100
+ "D209", # Closing triple quotes go to new line
101
+ "D213", # summary lines must be positioned on the second physical line of the docstring
102
+ "D417", # Missing argument descriptions in the docstring
103
+ "D419", # Docstring is empty
96
104
  "DTZ",
97
105
  "E",
98
106
  "EM",
@@ -144,14 +152,15 @@ ban-relative-imports = "parents"
144
152
 
145
153
  [tool.ruff.lint.per-file-ignores]
146
154
  # Tests can use magic values, assertions, and relative imports
147
- "tests/**/*" = ["PLR2004", "S101", "TID252", "ANN"]
155
+ "tests/**/*" = ["D", "PLR2004", "S101", "TID252", "ANN"]
148
156
  # examples can contain "print" commands
149
- "examples/**/*" = ["T201"]
157
+ "examples/**/*" = ["D", "T201"]
150
158
 
151
159
 
152
160
  [tool.coverage.run]
153
161
  source = ["haystack_integrations"]
154
162
  branch = true
163
+ relative_files = true
155
164
  parallel = false
156
165
 
157
166
 
@@ -163,3 +172,10 @@ omit = [
163
172
  ]
164
173
  show_missing = true
165
174
  exclude_lines = ["no cov", "if __name__ == .__main__.:", "if TYPE_CHECKING:"]
175
+
176
+ [tool.pytest.ini_options]
177
+ addopts = "--strict-markers"
178
+ markers = [
179
+ "integration: integration tests",
180
+ ]
181
+ log_cli = true
@@ -482,8 +482,9 @@ class QdrantSparseEmbeddingRetriever:
482
482
  @component
483
483
  class QdrantHybridRetriever:
484
484
  """
485
- A component for retrieving documents from an QdrantDocumentStore using both dense and sparse vectors
486
- and fusing the results using Reciprocal Rank Fusion.
485
+ A component for retrieving documents from a QdrantDocumentStore using both dense and sparse vectors.
486
+
487
+ Fuses the results using Reciprocal Rank Fusion.
487
488
 
488
489
  Usage example:
489
490
  ```python
@@ -18,6 +18,7 @@ def convert_haystack_documents_to_qdrant_points(
18
18
  *,
19
19
  use_sparse_embeddings: bool,
20
20
  ) -> list[rest.PointStruct]:
21
+ """Convert a list of Haystack Document objects to Qdrant PointStruct objects."""
21
22
  points = []
22
23
  for document in documents:
23
24
  payload = document.to_dict(flatten=False)
@@ -61,6 +62,7 @@ QdrantPoint = rest.ScoredPoint | rest.Record
61
62
 
62
63
 
63
64
  def convert_qdrant_point_to_haystack_document(point: QdrantPoint, use_sparse_embeddings: bool) -> Document:
65
+ """Convert a Qdrant ScoredPoint or Record to a Haystack Document object."""
64
66
  payload = point.payload or {}
65
67
  payload["score"] = point.score if hasattr(point, "score") else None
66
68
 
@@ -1,5 +1,6 @@
1
1
  import inspect
2
2
  from collections.abc import AsyncGenerator, Generator
3
+ from dataclasses import replace
3
4
  from itertools import islice
4
5
  from typing import Any, ClassVar, cast
5
6
 
@@ -54,8 +55,9 @@ def get_batches_from_generator(iterable: list, n: int) -> Generator:
54
55
 
55
56
  class QdrantDocumentStore:
56
57
  """
57
- A QdrantDocumentStore implementation that you can use with any Qdrant instance: in-memory, disk-persisted,
58
- Docker-based, and Qdrant Cloud Cluster deployments.
58
+ A QdrantDocumentStore implementation that you can use with any Qdrant instance.
59
+
60
+ Supports in-memory, disk-persisted, Docker-based, and Qdrant Cloud Cluster deployments.
59
61
 
60
62
  Usage example by creating an in-memory instance:
61
63
 
@@ -376,6 +378,7 @@ class QdrantDocumentStore:
376
378
  ) -> int:
377
379
  """
378
380
  Writes documents to Qdrant using the specified policy.
381
+
379
382
  The QdrantDocumentStore can handle duplicate documents based on the given policy.
380
383
  The available policies are:
381
384
  - `FAIL`: The operation will raise an error if any document already exists.
@@ -429,6 +432,7 @@ class QdrantDocumentStore:
429
432
  ) -> int:
430
433
  """
431
434
  Asynchronously writes documents to Qdrant using the specified policy.
435
+
432
436
  The QdrantDocumentStore can handle duplicate documents based on the given policy.
433
437
  The available policies are:
434
438
  - `FAIL`: The operation will raise an error if any document already exists.
@@ -1223,8 +1227,9 @@ class QdrantDocumentStore:
1223
1227
  self, filters: dict[str, Any], metadata_fields: list[str]
1224
1228
  ) -> dict[str, int]:
1225
1229
  """
1226
- Asynchronously returns the number of unique values for each specified metadata field among documents that
1227
- match the filters.
1230
+ Asynchronously returns the number of unique values for each specified metadata field among documents.
1231
+
1232
+ Only documents that match the filters are considered.
1228
1233
 
1229
1234
  :param filters: The filters to restrict the documents considered.
1230
1235
  For filter syntax, see [Haystack metadata filtering](https://docs.haystack.deepset.ai/docs/metadata-filtering)
@@ -1926,8 +1931,9 @@ class QdrantDocumentStore:
1926
1931
  group_size: int | None = None,
1927
1932
  ) -> list[Document]:
1928
1933
  """
1929
- Asynchronously retrieves documents based on dense and sparse embeddings and fuses
1930
- the results using Reciprocal Rank Fusion.
1934
+ Asynchronously retrieves documents based on dense and sparse embeddings.
1935
+
1936
+ Fuses the results using Reciprocal Rank Fusion.
1931
1937
 
1932
1938
  This method is not part of the public interface of `QdrantDocumentStore` and shouldn't be used directly.
1933
1939
  Use the `QdrantHybridRetriever` instead.
@@ -2292,8 +2298,9 @@ class QdrantDocumentStore:
2292
2298
  policy: DuplicatePolicy | None = None,
2293
2299
  ) -> list[Document]:
2294
2300
  """
2295
- Checks whether any of the passed documents is already existing in the chosen index and returns a list of
2296
- documents that are not in the index yet.
2301
+ Checks whether any of the passed documents is already existing in the chosen index.
2302
+
2303
+ Returns a list of documents that are not in the index yet.
2297
2304
 
2298
2305
  :param documents: A list of Haystack Document objects.
2299
2306
  :param policy: The duplicate policy to use when writing documents.
@@ -2319,9 +2326,9 @@ class QdrantDocumentStore:
2319
2326
  policy: DuplicatePolicy | None = None,
2320
2327
  ) -> list[Document]:
2321
2328
  """
2322
- Asynchronously checks whether any of the passed documents is already existing
2323
- in the chosen index and returns a list of
2324
- documents that are not in the index yet.
2329
+ Asynchronously checks whether any of the passed documents is already existing in the chosen index.
2330
+
2331
+ Returns a list of documents that are not in the index yet.
2325
2332
 
2326
2333
  :param documents: A list of Haystack Document objects.
2327
2334
  :param policy: The duplicate policy to use when writing documents.
@@ -2468,15 +2475,17 @@ class QdrantDocumentStore:
2468
2475
  ]
2469
2476
 
2470
2477
  if scale_score:
2471
- for document in documents:
2472
- score = document.score
2473
- if score is None:
2474
- continue
2475
- if self.similarity == "cosine":
2476
- score = (score + 1) / 2
2477
- else:
2478
- score = float(1 / (1 + exp(-score / 100)))
2479
- document.score = score
2478
+ documents = [
2479
+ replace(
2480
+ document,
2481
+ score=(document.score + 1) / 2
2482
+ if self.similarity == "cosine"
2483
+ else float(1 / (1 + exp(-document.score / 100))),
2484
+ )
2485
+ if document.score is not None
2486
+ else document
2487
+ for document in documents
2488
+ ]
2480
2489
 
2481
2490
  return documents
2482
2491
 
@@ -9,7 +9,8 @@ from qdrant_client.http import models
9
9
  def convert_filters_to_qdrant(
10
10
  filter_term: list[dict[str, Any]] | dict[str, Any] | models.Filter | None = None,
11
11
  ) -> models.Filter | None:
12
- """Converts Haystack filters to the format used by Qdrant.
12
+ """
13
+ Converts Haystack filters to the format used by Qdrant.
13
14
 
14
15
  :param filter_term: the haystack filter to be converted to qdrant.
15
16
  :returns: a single Qdrant Filter or None.
@@ -228,6 +229,7 @@ def _build_gte_condition(key: str, value: str | float | int) -> models.Condition
228
229
 
229
230
 
230
231
  def is_datetime_string(value: str) -> bool:
232
+ """Return True if the given string can be parsed as an ISO 8601 datetime, False otherwise."""
231
233
  try:
232
234
  datetime.fromisoformat(value)
233
235
  return True
@@ -0,0 +1,144 @@
1
+ from types import SimpleNamespace
2
+
3
+ import numpy as np
4
+ import pytest
5
+ from haystack.dataclasses import Document, SparseEmbedding
6
+ from qdrant_client.http import models as rest
7
+
8
+ from haystack_integrations.document_stores.qdrant.converters import (
9
+ DENSE_VECTORS_NAME,
10
+ SPARSE_VECTORS_NAME,
11
+ convert_haystack_documents_to_qdrant_points,
12
+ convert_id,
13
+ convert_qdrant_point_to_haystack_document,
14
+ )
15
+
16
+
17
+ def test_convert_id_is_deterministic():
18
+ first_id = convert_id("test-id")
19
+ second_id = convert_id("test-id")
20
+ assert first_id == second_id
21
+
22
+
23
+ def test_point_to_document_reverts_proper_structure_from_record_with_sparse():
24
+ point = rest.Record(
25
+ id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
26
+ payload={
27
+ "id": "my-id",
28
+ "id_hash_keys": ["content"],
29
+ "content": "Lorem ipsum",
30
+ "content_type": "text",
31
+ "meta": {
32
+ "test_field": 1,
33
+ },
34
+ },
35
+ vector={
36
+ "text-dense": [1.0, 0.0, 0.0, 0.0],
37
+ "text-sparse": {"indices": [7, 1024, 367], "values": [0.1, 0.98, 0.33]},
38
+ },
39
+ )
40
+ document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
41
+ assert "my-id" == document.id
42
+ assert "Lorem ipsum" == document.content
43
+ assert "text" == document.content_type
44
+ assert {"indices": [7, 1024, 367], "values": [0.1, 0.98, 0.33]} == document.sparse_embedding.to_dict()
45
+ assert {"test_field": 1} == document.meta
46
+ assert 0.0 == np.sum(np.array([1.0, 0.0, 0.0, 0.0]) - document.embedding)
47
+
48
+
49
+ def test_point_to_document_reverts_proper_structure_from_record_without_sparse():
50
+ point = rest.Record(
51
+ id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
52
+ payload={
53
+ "id": "my-id",
54
+ "id_hash_keys": ["content"],
55
+ "content": "Lorem ipsum",
56
+ "content_type": "text",
57
+ "meta": {
58
+ "test_field": 1,
59
+ },
60
+ },
61
+ vector=[1.0, 0.0, 0.0, 0.0],
62
+ )
63
+ document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=False)
64
+ assert "my-id" == document.id
65
+ assert "Lorem ipsum" == document.content
66
+ assert "text" == document.content_type
67
+ assert document.sparse_embedding is None
68
+ assert {"test_field": 1} == document.meta
69
+ assert 0.0 == np.sum(np.array([1.0, 0.0, 0.0, 0.0]) - document.embedding)
70
+
71
+
72
+ def test_point_to_document_with_sparse_enabled_but_vector_none():
73
+ point = rest.Record(
74
+ id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
75
+ payload={"id": "my-id", "content": "Lorem"},
76
+ vector=None,
77
+ )
78
+ document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
79
+ assert document.embedding is None
80
+ assert document.sparse_embedding is None
81
+
82
+
83
+ def test_point_to_document_preserves_score_from_scored_point():
84
+ point = rest.ScoredPoint(
85
+ id="c7c62e8e-02b9-4ec6-9f88-46bd97b628b7",
86
+ payload={"id": "my-id", "content": "Lorem"},
87
+ vector=[0.1, 0.2],
88
+ score=0.75,
89
+ version=0,
90
+ )
91
+ document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=False)
92
+ assert document.score == 0.75
93
+
94
+
95
+ def test_convert_haystack_documents_to_qdrant_points_without_sparse():
96
+ doc = Document(content="hello", embedding=[0.1, 0.2, 0.3])
97
+ points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=False)
98
+ assert len(points) == 1
99
+ assert points[0].vector == [0.1, 0.2, 0.3]
100
+ assert points[0].payload["content"] == "hello"
101
+ assert "embedding" not in points[0].payload
102
+
103
+
104
+ def test_convert_haystack_documents_to_qdrant_points_without_sparse_without_embedding():
105
+ doc = Document(content="hello")
106
+ points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=False)
107
+ assert points[0].vector == {}
108
+
109
+
110
+ def test_convert_haystack_documents_to_qdrant_points_with_sparse():
111
+ sparse = SparseEmbedding(indices=[0, 5], values=[0.1, 0.7])
112
+ doc = Document(content="hello", embedding=[0.1, 0.2], sparse_embedding=sparse)
113
+ points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
114
+ assert points[0].vector[DENSE_VECTORS_NAME] == [0.1, 0.2]
115
+ assert isinstance(points[0].vector[SPARSE_VECTORS_NAME], rest.SparseVector)
116
+ assert points[0].vector[SPARSE_VECTORS_NAME].indices == [0, 5]
117
+ assert points[0].vector[SPARSE_VECTORS_NAME].values == [0.1, 0.7]
118
+
119
+
120
+ def test_convert_haystack_documents_to_qdrant_points_with_sparse_only_dense():
121
+ doc = Document(content="hello", embedding=[0.1, 0.2])
122
+ points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
123
+ assert points[0].vector == {DENSE_VECTORS_NAME: [0.1, 0.2]}
124
+
125
+
126
+ def test_convert_haystack_documents_to_qdrant_points_with_sparse_no_vectors():
127
+ doc = Document(content="hello")
128
+ points = convert_haystack_documents_to_qdrant_points([doc], use_sparse_embeddings=True)
129
+ assert points[0].vector == {}
130
+
131
+
132
+ @pytest.mark.parametrize(
133
+ "vector",
134
+ [
135
+ {DENSE_VECTORS_NAME: [0.1, 0.2]},
136
+ {DENSE_VECTORS_NAME: [0.1, 0.2], SPARSE_VECTORS_NAME: {"indices": [0], "values": [0.5]}},
137
+ ],
138
+ ids=["no_sparse_key", "sparse_value_not_sparse_vector_instance"],
139
+ )
140
+ def test_point_to_document_sparse_vector_edge_cases(vector):
141
+ point = SimpleNamespace(id="x", payload={"id": "x", "content": "x"}, vector=vector)
142
+ document = convert_qdrant_point_to_haystack_document(point, use_sparse_embeddings=True)
143
+ assert document.embedding == [0.1, 0.2]
144
+ assert document.sparse_embedding is None