qdrant-haystack 10.2.0__tar.gz → 10.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/CHANGELOG.md +13 -0
  2. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/PKG-INFO +4 -3
  3. qdrant_haystack-10.3.0/pydoc/config_docusaurus.yml +15 -0
  4. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/pyproject.toml +11 -3
  5. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/document_stores/qdrant/document_store.py +119 -25
  6. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_document_store.py +23 -303
  7. qdrant_haystack-10.2.0/pydoc/config_docusaurus.yml +0 -30
  8. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/.gitignore +0 -0
  9. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/LICENSE.txt +0 -0
  10. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/README.md +0 -0
  11. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/examples/embedding_retrieval.py +0 -0
  12. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/components/retrievers/py.typed +0 -0
  13. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/components/retrievers/qdrant/__init__.py +0 -0
  14. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/components/retrievers/qdrant/retriever.py +0 -0
  15. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/document_stores/py.typed +0 -0
  16. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/document_stores/qdrant/__init__.py +0 -0
  17. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/document_stores/qdrant/converters.py +0 -0
  18. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/document_stores/qdrant/filters.py +0 -0
  19. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/src/haystack_integrations/document_stores/qdrant/migrate_to_sparse.py +0 -0
  20. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/__init__.py +0 -0
  21. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/conftest.py +0 -0
  22. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_converters.py +0 -0
  23. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_dict_converters.py +0 -0
  24. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_document_store_async.py +0 -0
  25. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_embedding_retriever.py +0 -0
  26. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_filters.py +0 -0
  27. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_hybrid_retriever.py +0 -0
  28. {qdrant_haystack-10.2.0 → qdrant_haystack-10.3.0}/tests/test_sparse_embedding_retriever.py +0 -0
@@ -1,5 +1,18 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/qdrant-v10.2.1] - 2026-02-02
4
+
5
+ ### 📚 Documentation
6
+
7
+ - Fixing `QdrantDocumentStore` docstring parsing error (#2806)
8
+
9
+
10
+ ## [integrations/qdrant-v10.2.0] - 2026-02-02
11
+
12
+ ### 🌀 Miscellaneous
13
+
14
+ - Feat: `QdrantDocumentStore` return number deleted docs on `delete_by_filter` (#2807)
15
+
3
16
  ## [integrations/qdrant-v9.6.0] - 2026-02-02
4
17
 
5
18
  ### 🚀 Features
@@ -1,11 +1,11 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: qdrant-haystack
3
- Version: 10.2.0
3
+ Version: 10.3.0
4
4
  Summary: An integration of Qdrant ANN vector database backend with Haystack
5
5
  Project-URL: Source, https://github.com/deepset-ai/haystack-core-integrations
6
6
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/blob/main/integrations/qdrant/README.md
7
7
  Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
8
- Author-email: Kacper Łukawski <kacper.lukawski@qdrant.com>, Anush Shetty <anush.shetty@qdrant.com>
8
+ Author-email: deepset GmbH <info@deepset.ai>, Kacper Łukawski <kacper.lukawski@qdrant.com>, Anush Shetty <anush.shetty@qdrant.com>
9
9
  License-Expression: Apache-2.0
10
10
  License-File: LICENSE.txt
11
11
  Classifier: Development Status :: 4 - Beta
@@ -15,10 +15,11 @@ Classifier: Programming Language :: Python :: 3.10
15
15
  Classifier: Programming Language :: Python :: 3.11
16
16
  Classifier: Programming Language :: Python :: 3.12
17
17
  Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
18
19
  Classifier: Programming Language :: Python :: Implementation :: CPython
19
20
  Classifier: Programming Language :: Python :: Implementation :: PyPy
20
21
  Requires-Python: >=3.10
21
- Requires-Dist: haystack-ai>=2.22.0
22
+ Requires-Dist: haystack-ai>=2.26.1
22
23
  Requires-Dist: qdrant-client>=1.12.0
23
24
  Description-Content-Type: text/markdown
24
25
 
@@ -0,0 +1,15 @@
1
+ loaders:
2
+ - modules:
3
+ - haystack_integrations.components.retrievers.qdrant.retriever
4
+ - haystack_integrations.document_stores.qdrant.document_store
5
+ - haystack_integrations.document_stores.qdrant.migrate_to_sparse
6
+ search_path: [../src]
7
+ processors:
8
+ - type: filter
9
+ documented_only: true
10
+ skip_empty_modules: true
11
+ renderer:
12
+ description: Qdrant integration for Haystack
13
+ id: integrations-qdrant
14
+ filename: qdrant.md
15
+ title: Qdrant
@@ -11,6 +11,7 @@ requires-python = ">=3.10"
11
11
  license = "Apache-2.0"
12
12
  keywords = []
13
13
  authors = [
14
+ { name = "deepset GmbH", email = "info@deepset.ai" },
14
15
  { name = "Kacper Łukawski", email = "kacper.lukawski@qdrant.com" },
15
16
  { name = "Anush Shetty", email = "anush.shetty@qdrant.com" },
16
17
  ]
@@ -22,10 +23,14 @@ classifiers = [
22
23
  "Programming Language :: Python :: 3.11",
23
24
  "Programming Language :: Python :: 3.12",
24
25
  "Programming Language :: Python :: 3.13",
26
+ "Programming Language :: Python :: 3.14",
25
27
  "Programming Language :: Python :: Implementation :: CPython",
26
28
  "Programming Language :: Python :: Implementation :: PyPy",
27
29
  ]
28
- dependencies = ["haystack-ai>=2.22.0", "qdrant-client>=1.12.0"]
30
+ dependencies = [
31
+ "haystack-ai>=2.26.1",
32
+ "qdrant-client>=1.12.0"
33
+ ]
29
34
 
30
35
  [project.urls]
31
36
  Source = "https://github.com/deepset-ai/haystack-core-integrations"
@@ -48,7 +53,7 @@ installer = "uv"
48
53
  dependencies = ["haystack-pydoc-tools", "ruff"]
49
54
 
50
55
  [tool.hatch.envs.default.scripts]
51
- docs = ["pydoc-markdown pydoc/config_docusaurus.yml"]
56
+ docs = ["haystack-pydoc pydoc/config_docusaurus.yml"]
52
57
  fmt = "ruff check --fix {args}; ruff format {args}"
53
58
  fmt-check = "ruff check {args} && ruff format --check {args}"
54
59
 
@@ -84,6 +89,7 @@ line-length = 120
84
89
  [tool.ruff.lint]
85
90
  select = [
86
91
  "A",
92
+ "ANN",
87
93
  "ARG",
88
94
  "B",
89
95
  "C",
@@ -112,6 +118,8 @@ select = [
112
118
  ignore = [
113
119
  # Allow non-abstract empty methods in abstract base classes
114
120
  "B027",
121
+ # Allow Any in type annotations at dynamic boundaries
122
+ "ANN401",
115
123
  # Allow boolean positional values in function calls, like `dict.get(... True)`
116
124
  "FBT003",
117
125
  # Allow boolean arguments in function definition
@@ -136,7 +144,7 @@ ban-relative-imports = "parents"
136
144
 
137
145
  [tool.ruff.lint.per-file-ignores]
138
146
  # Tests can use magic values, assertions, and relative imports
139
- "tests/**/*" = ["PLR2004", "S101", "TID252"]
147
+ "tests/**/*" = ["PLR2004", "S101", "TID252", "ANN"]
140
148
  # examples can contain "print" commands
141
149
  "examples/**/*" = ["T201"]
142
150
 
@@ -10,6 +10,7 @@ from haystack.dataclasses.sparse_embedding import SparseEmbedding
10
10
  from haystack.document_stores.errors import DocumentStoreError, DuplicateDocumentError
11
11
  from haystack.document_stores.types import DuplicatePolicy
12
12
  from haystack.utils import Secret, deserialize_secrets_inplace
13
+ from haystack.utils.misc import _normalize_metadata_field_name
13
14
  from numpy import exp
14
15
  from qdrant_client.http import models as rest
15
16
  from qdrant_client.http.exceptions import UnexpectedResponse
@@ -603,14 +604,59 @@ class QdrantDocumentStore:
603
604
  )
604
605
 
605
606
  @staticmethod
606
- def _metadata_fields_info_from_schema(payload_schema: dict[str, Any]) -> dict[str, str]:
607
- """Build field name -> type dict from Qdrant payload_schema. Used by get_metadata_fields_info (sync/async)."""
608
- fields_info: dict[str, str] = {}
607
+ def _infer_type_from_value(value: Any) -> str:
608
+ """
609
+ Infers the type from a metadata value for get_metadata_fields_info.
610
+
611
+ Returns types matching OpenSearch format for consistency:
612
+ - 'keyword' for strings
613
+ - 'long' for integers
614
+ - 'float' for floats
615
+ - 'boolean' for booleans
616
+ """
617
+ if isinstance(value, bool):
618
+ return "boolean"
619
+ elif isinstance(value, int):
620
+ return "long"
621
+ elif isinstance(value, float):
622
+ return "float"
623
+ elif isinstance(value, str):
624
+ return "keyword"
625
+ else:
626
+ return "keyword"
627
+
628
+ @staticmethod
629
+ def _process_records_fields_info(records: list[Any], field_info: dict[str, dict[str, str]]) -> None:
630
+ """
631
+ Update field_info from a batch of Qdrant records.
632
+
633
+ Used by get_metadata_fields_info (sync/async). Extracts metadata from
634
+ payload["meta"] and infers types for each field.
635
+ """
636
+ for record in records:
637
+ if record.payload and "meta" in record.payload:
638
+ meta = record.payload["meta"]
639
+ for field_name, value in meta.items():
640
+ if value is not None and field_name not in field_info:
641
+ field_info[field_name] = {"type": QdrantDocumentStore._infer_type_from_value(value)}
642
+
643
+ @staticmethod
644
+ def _metadata_fields_info_from_schema(payload_schema: dict[str, Any]) -> dict[str, dict[str, str]]:
645
+ """
646
+ Build field name -> {type: ...} dict from Qdrant payload_schema.
647
+
648
+ Used when payload_schema has indexed metadata fields (e.g. meta.category).
649
+ Returns empty dict when schema has no metadata field info.
650
+ """
651
+ fields_info: dict[str, dict[str, str]] = {}
609
652
  for field_name, field_config in payload_schema.items():
610
- if hasattr(field_config, "data_type"):
611
- fields_info[field_name] = str(field_config.data_type)
612
- else:
613
- fields_info[field_name] = "unknown"
653
+ if field_name.startswith("meta."):
654
+ meta_field = field_name[5:]
655
+ if hasattr(field_config, "data_type"):
656
+ qdrant_type = str(field_config.data_type).lower()
657
+ fields_info[meta_field] = {"type": qdrant_type}
658
+ else:
659
+ fields_info[meta_field] = {"type": "unknown"}
614
660
  return fields_info
615
661
 
616
662
  @staticmethod
@@ -978,12 +1024,19 @@ class QdrantDocumentStore:
978
1024
  logger.warning(f"Error {e} when calling QdrantDocumentStore.count_documents_by_filter_async()")
979
1025
  return 0
980
1026
 
981
- def get_metadata_fields_info(self) -> dict[str, str]:
1027
+ def get_metadata_fields_info(self) -> dict[str, dict[str, str]]:
982
1028
  """
983
- Returns the information about the fields from the collection.
1029
+ Returns the information about the metadata fields in the collection.
1030
+
1031
+ Since Qdrant may not have a payload schema for unindexed metadata,
1032
+ this method scrolls through documents to infer field types from
1033
+ payload["meta"].
984
1034
 
985
1035
  :returns:
986
- A dictionary mapping field names to their types (e.g., {"field_name": "integer"}).
1036
+ A dictionary mapping field names to their type information e.g.:
1037
+ ```python
1038
+ {"category": {"type": "keyword"}, "priority": {"type": "long"}}
1039
+ ```
987
1040
  """
988
1041
  self._initialize_client()
989
1042
  assert self._client is not None
@@ -991,17 +1044,41 @@ class QdrantDocumentStore:
991
1044
  try:
992
1045
  collection_info = self._client.get_collection(self.index)
993
1046
  payload_schema = collection_info.payload_schema or {}
994
- return self._metadata_fields_info_from_schema(payload_schema)
1047
+ fields_info = self._metadata_fields_info_from_schema(payload_schema)
1048
+
1049
+ if not fields_info:
1050
+ next_offset = None
1051
+ while True:
1052
+ records, next_offset = self._client.scroll(
1053
+ collection_name=self.index,
1054
+ scroll_filter=None,
1055
+ limit=self.scroll_size,
1056
+ offset=next_offset,
1057
+ with_payload=True,
1058
+ with_vectors=False,
1059
+ )
1060
+ self._process_records_fields_info(records, fields_info)
1061
+ if self._check_stop_scrolling(next_offset):
1062
+ break
1063
+
1064
+ return fields_info
995
1065
  except (UnexpectedResponse, ValueError) as e:
996
1066
  logger.warning(f"Error {e} when calling QdrantDocumentStore.get_metadata_fields_info()")
997
1067
  return {}
998
1068
 
999
- async def get_metadata_fields_info_async(self) -> dict[str, str]:
1069
+ async def get_metadata_fields_info_async(self) -> dict[str, dict[str, str]]:
1000
1070
  """
1001
- Asynchronously returns the information about the fields from the collection.
1071
+ Asynchronously returns the information about the metadata fields in the collection.
1072
+
1073
+ Since Qdrant may not have a payload schema for unindexed metadata,
1074
+ this method scrolls through documents to infer field types from
1075
+ payload["meta"].
1002
1076
 
1003
1077
  :returns:
1004
- A dictionary mapping field names to their types (e.g., {"field_name": "integer"}).
1078
+ A dictionary mapping field names to their type information e.g.:
1079
+ ```python
1080
+ {"category": {"type": "keyword"}, "priority": {"type": "long"}}
1081
+ ```
1005
1082
  """
1006
1083
  await self._initialize_async_client()
1007
1084
  assert self._async_client is not None
@@ -1009,7 +1086,24 @@ class QdrantDocumentStore:
1009
1086
  try:
1010
1087
  collection_info = await self._async_client.get_collection(self.index)
1011
1088
  payload_schema = collection_info.payload_schema or {}
1012
- return self._metadata_fields_info_from_schema(payload_schema)
1089
+ fields_info = self._metadata_fields_info_from_schema(payload_schema)
1090
+
1091
+ if not fields_info:
1092
+ next_offset = None
1093
+ while True:
1094
+ records, next_offset = await self._async_client.scroll(
1095
+ collection_name=self.index,
1096
+ scroll_filter=None,
1097
+ limit=self.scroll_size,
1098
+ offset=next_offset,
1099
+ with_payload=True,
1100
+ with_vectors=False,
1101
+ )
1102
+ self._process_records_fields_info(records, fields_info)
1103
+ if self._check_stop_scrolling(next_offset):
1104
+ break
1105
+
1106
+ return fields_info
1013
1107
  except (UnexpectedResponse, ValueError) as e:
1014
1108
  logger.warning(f"Error {e} when calling QdrantDocumentStore.get_metadata_fields_info_async()")
1015
1109
  return {}
@@ -1021,11 +1115,13 @@ class QdrantDocumentStore:
1021
1115
  :param metadata_field: The metadata field key (inside ``meta``) to get the minimum and maximum values for.
1022
1116
 
1023
1117
  :returns: A dictionary with the keys "min" and "max", where each value is the minimum or maximum value of the
1024
- metadata field across all documents. Returns an empty dict if no documents have the field.
1118
+ metadata field across all documents. Returns ``{"min": None, "max": None}`` if no documents have
1119
+ the field.
1025
1120
  """
1026
1121
  self._initialize_client()
1027
1122
  assert self._client is not None
1028
1123
 
1124
+ field_name = _normalize_metadata_field_name(metadata_field)
1029
1125
  try:
1030
1126
  min_value: Any = None
1031
1127
  max_value: Any = None
@@ -1040,13 +1136,11 @@ class QdrantDocumentStore:
1040
1136
  with_payload=True,
1041
1137
  with_vectors=False,
1042
1138
  )
1043
- min_value, max_value = self._process_records_min_max(records, metadata_field, min_value, max_value)
1139
+ min_value, max_value = self._process_records_min_max(records, field_name, min_value, max_value)
1044
1140
  if self._check_stop_scrolling(next_offset):
1045
1141
  break
1046
1142
 
1047
- if min_value is not None and max_value is not None:
1048
- return {"min": min_value, "max": max_value}
1049
- return {}
1143
+ return {"min": min_value, "max": max_value}
1050
1144
  except Exception as e:
1051
1145
  logger.warning(f"Error {e} when calling QdrantDocumentStore.get_metadata_field_min_max()")
1052
1146
  return {}
@@ -1058,11 +1152,13 @@ class QdrantDocumentStore:
1058
1152
  :param metadata_field: The metadata field key (inside ``meta``) to get the minimum and maximum values for.
1059
1153
 
1060
1154
  :returns: A dictionary with the keys "min" and "max", where each value is the minimum or maximum value of the
1061
- metadata field across all documents. Returns an empty dict if no documents have the field.
1155
+ metadata field across all documents. Returns ``{"min": None, "max": None}`` if no documents have
1156
+ the field.
1062
1157
  """
1063
1158
  await self._initialize_async_client()
1064
1159
  assert self._async_client is not None
1065
1160
 
1161
+ field_name = _normalize_metadata_field_name(metadata_field)
1066
1162
  try:
1067
1163
  min_value: Any = None
1068
1164
  max_value: Any = None
@@ -1077,13 +1173,11 @@ class QdrantDocumentStore:
1077
1173
  with_payload=True,
1078
1174
  with_vectors=False,
1079
1175
  )
1080
- min_value, max_value = self._process_records_min_max(records, metadata_field, min_value, max_value)
1176
+ min_value, max_value = self._process_records_min_max(records, field_name, min_value, max_value)
1081
1177
  if self._check_stop_scrolling(next_offset):
1082
1178
  break
1083
1179
 
1084
- if min_value is not None and max_value is not None:
1085
- return {"min": min_value, "max": max_value}
1086
- return {}
1180
+ return {"min": min_value, "max": max_value}
1087
1181
  except Exception as e:
1088
1182
  logger.warning(f"Error {e} when calling QdrantDocumentStore.get_metadata_field_min_max_async()")
1089
1183
  return {}
@@ -6,8 +6,17 @@ from haystack.dataclasses import SparseEmbedding
6
6
  from haystack.document_stores.errors import DuplicateDocumentError
7
7
  from haystack.document_stores.types import DuplicatePolicy
8
8
  from haystack.testing.document_store import (
9
+ CountDocumentsByFilterTest,
9
10
  CountDocumentsTest,
11
+ CountUniqueMetadataByFilterTest,
12
+ DeleteAllTest,
13
+ DeleteByFilterTest,
10
14
  DeleteDocumentsTest,
15
+ FilterableDocsFixtureMixin,
16
+ GetMetadataFieldMinMaxTest,
17
+ GetMetadataFieldsInfoTest,
18
+ GetMetadataFieldUniqueValuesTest,
19
+ UpdateByFilterTest,
11
20
  WriteDocumentsTest,
12
21
  _random_embeddings,
13
22
  )
@@ -22,7 +31,20 @@ from haystack_integrations.document_stores.qdrant.document_store import (
22
31
  )
23
32
 
24
33
 
25
- class TestQdrantDocumentStore(CountDocumentsTest, WriteDocumentsTest, DeleteDocumentsTest):
34
+ class TestQdrantDocumentStore(
35
+ CountDocumentsByFilterTest,
36
+ CountDocumentsTest,
37
+ CountUniqueMetadataByFilterTest,
38
+ DeleteAllTest,
39
+ DeleteByFilterTest,
40
+ DeleteDocumentsTest,
41
+ FilterableDocsFixtureMixin,
42
+ GetMetadataFieldMinMaxTest,
43
+ GetMetadataFieldUniqueValuesTest,
44
+ GetMetadataFieldsInfoTest,
45
+ UpdateByFilterTest,
46
+ WriteDocumentsTest,
47
+ ):
26
48
  @pytest.fixture
27
49
  def document_store(self) -> QdrantDocumentStore:
28
50
  return QdrantDocumentStore(
@@ -302,21 +324,6 @@ class TestQdrantDocumentStore(CountDocumentsTest, WriteDocumentsTest, DeleteDocu
302
324
  with pytest.raises(ValueError, match="different vector size"):
303
325
  document_store._set_up_collection("test_collection", 768, False, "cosine", False, False)
304
326
 
305
- def test_delete_all_documents_no_index_recreation(self, document_store):
306
- document_store._initialize_client()
307
-
308
- # write some documents
309
- docs = [Document(id=str(i)) for i in range(5)]
310
- document_store.write_documents(docs)
311
-
312
- # delete all documents without recreating the index
313
- document_store.delete_all_documents(recreate_index=False)
314
- assert document_store.count_documents() == 0
315
-
316
- # ensure the collection still exists by writing documents again
317
- document_store.write_documents(docs)
318
- assert document_store.count_documents() == 5
319
-
320
327
  def test_delete_all_documents_index_recreation(self, document_store):
321
328
  document_store._initialize_client()
322
329
 
@@ -340,183 +347,6 @@ class TestQdrantDocumentStore(CountDocumentsTest, WriteDocumentsTest, DeleteDocu
340
347
  document_store.write_documents(docs)
341
348
  assert document_store.count_documents() == 5
342
349
 
343
- def test_delete_by_filter(self, document_store: QdrantDocumentStore):
344
- docs = [
345
- Document(content="Doc 1", meta={"category": "A", "year": 2023}),
346
- Document(content="Doc 2", meta={"category": "B", "year": 2023}),
347
- Document(content="Doc 3", meta={"category": "A", "year": 2024}),
348
- ]
349
- document_store.write_documents(docs)
350
- assert document_store.count_documents() == 3
351
-
352
- deleted_count = document_store.delete_by_filter(
353
- filters={"field": "meta.category", "operator": "==", "value": "A"}
354
- )
355
- assert deleted_count == 2
356
-
357
- # Verify only category B remains
358
- remaining_docs = document_store.filter_documents()
359
- assert len(remaining_docs) == 1
360
- assert remaining_docs[0].meta["category"] == "B"
361
-
362
- # Delete remaining document by year
363
- deleted_count = document_store.delete_by_filter(filters={"field": "meta.year", "operator": "==", "value": 2023})
364
- assert deleted_count == 1
365
- assert document_store.count_documents() == 0
366
-
367
- def test_delete_by_filter_no_matches(self, document_store: QdrantDocumentStore):
368
- docs = [
369
- Document(content="Doc 1", meta={"category": "A"}),
370
- Document(content="Doc 2", meta={"category": "B"}),
371
- ]
372
- document_store.write_documents(docs)
373
- assert document_store.count_documents() == 2
374
-
375
- # try to delete documents with category="C" (no matches)
376
- deleted_count = document_store.delete_by_filter(
377
- filters={"field": "meta.category", "operator": "==", "value": "C"}
378
- )
379
- assert deleted_count == 0
380
- assert document_store.count_documents() == 2
381
-
382
- def test_delete_by_filter_advanced_filters(self, document_store: QdrantDocumentStore):
383
- docs = [
384
- Document(content="Doc 1", meta={"category": "A", "year": 2023, "status": "draft"}),
385
- Document(content="Doc 2", meta={"category": "A", "year": 2024, "status": "published"}),
386
- Document(content="Doc 3", meta={"category": "B", "year": 2023, "status": "draft"}),
387
- ]
388
- document_store.write_documents(docs)
389
- assert document_store.count_documents() == 3
390
-
391
- # AND condition (matches only Doc 1)
392
- deleted_count = document_store.delete_by_filter(
393
- filters={
394
- "operator": "AND",
395
- "conditions": [
396
- {"field": "meta.category", "operator": "==", "value": "A"},
397
- {"field": "meta.year", "operator": "==", "value": 2023},
398
- ],
399
- }
400
- )
401
- assert deleted_count == 1
402
- assert document_store.count_documents() == 2
403
-
404
- # OR condition (matches Doc 2 and Doc 3)
405
- deleted_count = document_store.delete_by_filter(
406
- filters={
407
- "operator": "OR",
408
- "conditions": [
409
- {"field": "meta.category", "operator": "==", "value": "B"},
410
- {"field": "meta.status", "operator": "==", "value": "published"},
411
- ],
412
- }
413
- )
414
- assert deleted_count == 2
415
- assert document_store.count_documents() == 0
416
-
417
- def test_update_by_filter(self, document_store: QdrantDocumentStore):
418
- docs = [
419
- Document(content="Doc 1", meta={"category": "A", "status": "draft"}),
420
- Document(content="Doc 2", meta={"category": "B", "status": "draft"}),
421
- Document(content="Doc 3", meta={"category": "A", "status": "draft"}),
422
- ]
423
- document_store.write_documents(docs)
424
- assert document_store.count_documents() == 3
425
-
426
- # Update status for category="A" documents
427
- updated_count = document_store.update_by_filter(
428
- filters={"field": "meta.category", "operator": "==", "value": "A"}, meta={"status": "published"}
429
- )
430
- assert updated_count == 2
431
-
432
- # Verify the updated documents have the new metadata
433
- published_docs = document_store.filter_documents(
434
- filters={"field": "meta.status", "operator": "==", "value": "published"}
435
- )
436
- assert len(published_docs) == 2
437
- for doc in published_docs:
438
- assert doc.meta["status"] == "published"
439
- assert doc.meta["category"] == "A"
440
-
441
- # Verify documents with category="B" were not updated
442
- draft_docs = document_store.filter_documents(
443
- filters={"field": "meta.status", "operator": "==", "value": "draft"}
444
- )
445
- assert len(draft_docs) == 1
446
- assert draft_docs[0].meta["category"] == "B"
447
-
448
- def test_update_by_filter_multiple_fields(self, document_store: QdrantDocumentStore):
449
- docs = [
450
- Document(content="Doc 1", meta={"category": "A", "year": 2023}),
451
- Document(content="Doc 2", meta={"category": "A", "year": 2023}),
452
- Document(content="Doc 3", meta={"category": "B", "year": 2024}),
453
- ]
454
- document_store.write_documents(docs)
455
- assert document_store.count_documents() == 3
456
-
457
- # Update multiple fields for category="A" documents
458
- updated_count = document_store.update_by_filter(
459
- filters={"field": "meta.category", "operator": "==", "value": "A"},
460
- meta={"status": "published", "reviewed": True},
461
- )
462
- assert updated_count == 2
463
-
464
- # Verify updates
465
- published_docs = document_store.filter_documents(
466
- filters={"field": "meta.status", "operator": "==", "value": "published"}
467
- )
468
- assert len(published_docs) == 2
469
- for doc in published_docs:
470
- assert doc.meta["status"] == "published"
471
- assert doc.meta["reviewed"] is True
472
- assert doc.meta["category"] == "A"
473
- assert doc.meta["year"] == 2023 # Existing field preserved
474
-
475
- def test_update_by_filter_no_matches(self, document_store: QdrantDocumentStore):
476
- docs = [
477
- Document(content="Doc 1", meta={"category": "A"}),
478
- Document(content="Doc 2", meta={"category": "B"}),
479
- ]
480
- document_store.write_documents(docs)
481
- assert document_store.count_documents() == 2
482
-
483
- # Try to update documents with category="C" (no matches)
484
- updated_count = document_store.update_by_filter(
485
- filters={"field": "meta.category", "operator": "==", "value": "C"}, meta={"status": "published"}
486
- )
487
- assert updated_count == 0
488
- assert document_store.count_documents() == 2
489
-
490
- def test_update_by_filter_advanced_filters(self, document_store: QdrantDocumentStore):
491
- docs = [
492
- Document(content="Doc 1", meta={"category": "A", "year": 2023, "status": "draft"}),
493
- Document(content="Doc 2", meta={"category": "A", "year": 2024, "status": "draft"}),
494
- Document(content="Doc 3", meta={"category": "B", "year": 2023, "status": "draft"}),
495
- ]
496
- document_store.write_documents(docs)
497
- assert document_store.count_documents() == 3
498
-
499
- # Update with AND condition
500
- updated_count = document_store.update_by_filter(
501
- filters={
502
- "operator": "AND",
503
- "conditions": [
504
- {"field": "meta.category", "operator": "==", "value": "A"},
505
- {"field": "meta.year", "operator": "==", "value": 2023},
506
- ],
507
- },
508
- meta={"status": "published"},
509
- )
510
- assert updated_count == 1
511
-
512
- # Verify only one document was updated
513
- published_docs = document_store.filter_documents(
514
- filters={"field": "meta.status", "operator": "==", "value": "published"}
515
- )
516
- assert len(published_docs) == 1
517
- assert published_docs[0].meta["category"] == "A"
518
- assert published_docs[0].meta["year"] == 2023
519
-
520
350
  def test_update_by_filter_preserves_vectors(self, document_store: QdrantDocumentStore):
521
351
  """Test that update_by_filter preserves document embeddings."""
522
352
  docs = [
@@ -539,116 +369,6 @@ class TestQdrantDocumentStore(CountDocumentsTest, WriteDocumentsTest, DeleteDocu
539
369
  assert updated_docs[0].embedding is not None
540
370
  assert len(updated_docs[0].embedding) == 768
541
371
 
542
- def test_count_documents_by_filter(self, document_store: QdrantDocumentStore):
543
- """Test counting documents with filters."""
544
- docs = [
545
- Document(content="Doc 1", meta={"category": "A", "year": 2023}),
546
- Document(content="Doc 2", meta={"category": "A", "year": 2024}),
547
- Document(content="Doc 3", meta={"category": "B", "year": 2023}),
548
- Document(content="Doc 4", meta={"category": "B", "year": 2024}),
549
- ]
550
- document_store.write_documents(docs)
551
-
552
- # Test counting all documents
553
- assert document_store.count_documents() == 4
554
-
555
- # Test counting with single filter
556
- count = document_store.count_documents_by_filter(
557
- filters={"field": "meta.category", "operator": "==", "value": "A"}
558
- )
559
- assert count == 2
560
-
561
- # Test counting with multiple filters
562
- count = document_store.count_documents_by_filter(
563
- filters={
564
- "operator": "AND",
565
- "conditions": [
566
- {"field": "meta.category", "operator": "==", "value": "B"},
567
- {"field": "meta.year", "operator": "==", "value": 2023},
568
- ],
569
- }
570
- )
571
- assert count == 1
572
-
573
- def test_get_metadata_fields_info(self, document_store: QdrantDocumentStore):
574
- """Test getting metadata field information."""
575
- docs = [
576
- Document(content="Doc 1", meta={"category": "A", "score": 0.9, "tags": ["tag1", "tag2"]}),
577
- Document(content="Doc 2", meta={"category": "B", "score": 0.8, "tags": ["tag2"]}),
578
- ]
579
- document_store.write_documents(docs)
580
-
581
- fields_info = document_store.get_metadata_fields_info()
582
- # Should return empty dict or field info depending on Qdrant collection setup
583
- assert isinstance(fields_info, dict)
584
-
585
- def test_get_metadata_field_min_max(self, document_store: QdrantDocumentStore):
586
- """Test getting min/max values for a metadata field."""
587
- docs = [
588
- Document(content="Doc 1", meta={"score": 0.5}),
589
- Document(content="Doc 2", meta={"score": 0.8}),
590
- Document(content="Doc 3", meta={"score": 0.3}),
591
- ]
592
- document_store.write_documents(docs)
593
-
594
- result = document_store.get_metadata_field_min_max("score")
595
- assert result.get("min") == 0.3
596
- assert result.get("max") == 0.8
597
-
598
- def test_count_unique_metadata_by_filter(self, document_store: QdrantDocumentStore):
599
- """Test counting unique metadata field values."""
600
- docs = [
601
- Document(content="Doc 1", meta={"category": "A"}),
602
- Document(content="Doc 2", meta={"category": "B"}),
603
- Document(content="Doc 3", meta={"category": "A"}),
604
- Document(content="Doc 4", meta={"category": "C"}),
605
- ]
606
- document_store.write_documents(docs)
607
-
608
- result = document_store.count_unique_metadata_by_filter(filters={}, metadata_fields=["category"])
609
- assert result == {"category": 3}
610
-
611
- def test_count_unique_metadata_by_filter_multiple_fields(self, document_store: QdrantDocumentStore):
612
- """Test counting unique values for multiple metadata fields."""
613
- docs = [
614
- Document(content="Doc 1", meta={"category": "A", "status": "active"}),
615
- Document(content="Doc 2", meta={"category": "B", "status": "active"}),
616
- Document(content="Doc 3", meta={"category": "A", "status": "inactive"}),
617
- ]
618
- document_store.write_documents(docs)
619
-
620
- result = document_store.count_unique_metadata_by_filter(filters={}, metadata_fields=["category", "status"])
621
- assert result == {"category": 2, "status": 2}
622
-
623
- def test_count_unique_metadata_by_filter_with_filter(self, document_store: QdrantDocumentStore):
624
- """Test counting unique metadata field values with filtering."""
625
- docs = [
626
- Document(content="Doc 1", meta={"category": "A", "status": "active"}),
627
- Document(content="Doc 2", meta={"category": "B", "status": "active"}),
628
- Document(content="Doc 3", meta={"category": "A", "status": "inactive"}),
629
- ]
630
- document_store.write_documents(docs)
631
-
632
- result = document_store.count_unique_metadata_by_filter(
633
- filters={"field": "meta.status", "operator": "==", "value": "active"},
634
- metadata_fields=["category"],
635
- )
636
- assert result == {"category": 2}
637
-
638
- def test_get_metadata_field_unique_values(self, document_store: QdrantDocumentStore):
639
- """Test getting unique metadata field values."""
640
- docs = [
641
- Document(content="Doc 1", meta={"category": "A"}),
642
- Document(content="Doc 2", meta={"category": "B"}),
643
- Document(content="Doc 3", meta={"category": "A"}),
644
- Document(content="Doc 4", meta={"category": "C"}),
645
- ]
646
- document_store.write_documents(docs)
647
-
648
- values = document_store.get_metadata_field_unique_values("category")
649
- assert len(values) == 3
650
- assert set(values) == {"A", "B", "C"}
651
-
652
372
  def test_get_metadata_field_unique_values_pagination(self, document_store: QdrantDocumentStore):
653
373
  """Test getting unique metadata field values with pagination."""
654
374
  docs = [Document(content=f"Doc {i}", meta={"value": i % 5}) for i in range(10)]
@@ -1,30 +0,0 @@
1
- loaders:
2
- - ignore_when_discovered:
3
- - __init__
4
- modules:
5
- - haystack_integrations.components.retrievers.qdrant.retriever
6
- - haystack_integrations.document_stores.qdrant.document_store
7
- - haystack_integrations.document_stores.qdrant.migrate_to_sparse
8
- search_path:
9
- - ../src
10
- type: haystack_pydoc_tools.loaders.CustomPythonLoader
11
- processors:
12
- - do_not_filter_modules: false
13
- documented_only: true
14
- expression: null
15
- skip_empty_modules: true
16
- type: filter
17
- - type: smart
18
- - type: crossref
19
- renderer:
20
- description: Qdrant integration for Haystack
21
- id: integrations-qdrant
22
- markdown:
23
- add_member_class_prefix: false
24
- add_method_class_prefix: true
25
- classdef_code_block: false
26
- descriptive_class_title: false
27
- descriptive_module_title: true
28
- filename: qdrant.md
29
- title: Qdrant
30
- type: haystack_pydoc_tools.renderers.DocusaurusRenderer