chroma-haystack 4.3.2__tar.gz → 4.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/CHANGELOG.md +18 -0
  2. chroma_haystack-4.3.2/LICENSE → chroma_haystack-4.5.0/LICENSE.txt +1 -1
  3. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/PKG-INFO +5 -5
  4. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/README.md +2 -2
  5. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/pyproject.toml +1 -0
  6. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/components/retrievers/chroma/retriever.py +12 -0
  7. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/document_store.py +59 -34
  8. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_document_store.py +37 -60
  9. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_document_store_async.py +1 -11
  10. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_retriever.py +18 -0
  11. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/.gitignore +0 -0
  12. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_01.txt +0 -0
  13. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_02.txt +0 -0
  14. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_03.txt +0 -0
  15. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_04.txt +0 -0
  16. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_05.txt +0 -0
  17. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_06.txt +0 -0
  18. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_07.txt +0 -0
  19. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_08.txt +0 -0
  20. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_09.txt +0 -0
  21. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_10.txt +0 -0
  22. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_11.txt +0 -0
  23. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_12.txt +0 -0
  24. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_20.txt +0 -0
  25. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_21.txt +0 -0
  26. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_22.txt +0 -0
  27. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_23.txt +0 -0
  28. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_24.txt +0 -0
  29. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_25.txt +0 -0
  30. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_26.txt +0 -0
  31. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_27.txt +0 -0
  32. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_28.txt +0 -0
  33. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_29.txt +0 -0
  34. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_30.txt +0 -0
  35. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_31.txt +0 -0
  36. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_32.txt +0 -0
  37. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_40.txt +0 -0
  38. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_41.txt +0 -0
  39. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_42.txt +0 -0
  40. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_43.txt +0 -0
  41. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_44.txt +0 -0
  42. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_45.txt +0 -0
  43. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_46.txt +0 -0
  44. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_50.txt +0 -0
  45. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_51.txt +0 -0
  46. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_52.txt +0 -0
  47. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_90.txt +0 -0
  48. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/example.py +0 -0
  49. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/pydoc/config_docusaurus.yml +0 -0
  50. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/components/retrievers/chroma/__init__.py +0 -0
  51. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/components/retrievers/py.typed +0 -0
  52. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/__init__.py +0 -0
  53. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/errors.py +0 -0
  54. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/filters.py +0 -0
  55. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/utils.py +0 -0
  56. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/py.typed +0 -0
  57. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/__init__.py +0 -0
  58. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/conftest.py +0 -0
  59. {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_filters.py +0 -0
@@ -1,5 +1,23 @@
1
1
  # Changelog
2
2
 
3
+ ## [integrations/chroma-v4.4.0] - 2026-07-24
4
+
5
+ ### 🚀 Features
6
+
7
+ - Chroma - add sync closing methods (#3660)
8
+
9
+ ### 🐛 Bug Fixes
10
+
11
+ - `ChromaDocumentStore` match `search_term` against metadata field value, not content (#3644)
12
+
13
+
14
+ ## [integrations/chroma-v4.3.2] - 2026-07-17
15
+
16
+ ### 🐛 Bug Fixes
17
+
18
+ - *(chroma)* Raise a clear error for id filters with a non-'==' operator (#3587)
19
+
20
+
3
21
  ## [integrations/chroma-v4.3.1] - 2026-06-29
4
22
 
5
23
  ### 🐛 Bug Fixes
@@ -186,7 +186,7 @@
186
186
  same "printed page" as the copyright notice for easier
187
187
  identification within third-party archives.
188
188
 
189
- Copyright 2023 deepset GmbH
189
+ Copyright 2023-present deepset GmbH
190
190
 
191
191
  Licensed under the Apache License, Version 2.0 (the "License");
192
192
  you may not use this file except in compliance with the License.
@@ -1,12 +1,12 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: chroma-haystack
3
- Version: 4.3.2
3
+ Version: 4.5.0
4
4
  Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/chroma#readme
5
5
  Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
6
6
  Project-URL: Source, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/chroma
7
7
  Author-email: deepset GmbH <info@deepset.ai>
8
8
  License-Expression: Apache-2.0
9
- License-File: LICENSE
9
+ License-File: LICENSE.txt
10
10
  Classifier: Development Status :: 4 - Beta
11
11
  Classifier: License :: OSI Approved :: Apache Software License
12
12
  Classifier: Programming Language :: Python
@@ -36,5 +36,5 @@ Description-Content-Type: text/markdown
36
36
 
37
37
  Refer to the general [Contribution Guidelines](https://github.com/deepset-ai/haystack-core-integrations/blob/main/CONTRIBUTING.md).
38
38
 
39
- To run integration tests locally, you need a Chroma server running.
40
- Start one with: `docker run -p 8000:8000 chromadb/chroma:latest`.
39
+ To run integration tests locally, you need a Chroma server running on `localhost:8000`.
40
+ You can start it with: `hatch run chroma run`.
@@ -12,5 +12,5 @@
12
12
 
13
13
  Refer to the general [Contribution Guidelines](https://github.com/deepset-ai/haystack-core-integrations/blob/main/CONTRIBUTING.md).
14
14
 
15
- To run integration tests locally, you need a Chroma server running.
16
- Start one with: `docker run -p 8000:8000 chromadb/chroma:latest`.
15
+ To run integration tests locally, you need a Chroma server running on `localhost:8000`.
16
+ You can start it with: `hatch run chroma run`.
@@ -135,6 +135,7 @@ ignore = [
135
135
  "PLR0912",
136
136
  "PLR0913",
137
137
  "PLR0915",
138
+ "PLR0917",
138
139
  # Ignore unused params
139
140
  "ARG002",
140
141
  # Allow assertions
@@ -119,6 +119,12 @@ class ChromaQueryTextRetriever:
119
119
  top_k = top_k or self.top_k
120
120
  return {"documents": (await self.document_store.search_async([query], top_k, filters))[0]}
121
121
 
122
+ def close(self) -> None:
123
+ """
124
+ Release the synchronous resources of the underlying Document Store.
125
+ """
126
+ self.document_store.close()
127
+
122
128
  @classmethod
123
129
  def from_dict(cls, data: dict[str, Any]) -> "ChromaQueryTextRetriever":
124
130
  """
@@ -238,6 +244,12 @@ class ChromaEmbeddingRetriever:
238
244
  query_embeddings = [query_embedding]
239
245
  return {"documents": (await self.document_store.search_embeddings_async(query_embeddings, top_k, filters))[0]}
240
246
 
247
+ def close(self) -> None:
248
+ """
249
+ Release the synchronous resources of the underlying Document Store.
250
+ """
251
+ self.document_store.close()
252
+
241
253
  @classmethod
242
254
  def from_dict(cls, data: dict[str, Any]) -> "ChromaEmbeddingRetriever":
243
255
  """
@@ -3,9 +3,11 @@
3
3
  # SPDX-License-Identifier: Apache-2.0
4
4
 
5
5
  from collections.abc import Sequence
6
+ from contextlib import suppress
6
7
  from typing import Any, Literal, cast
7
8
 
8
9
  import chromadb
10
+ from chromadb.api import ClientAPI
9
11
  from chromadb.api.models.AsyncCollection import AsyncCollection
10
12
  from chromadb.api.types import GetResult, Metadata, OneOrMany, QueryResult
11
13
  from chromadb.config import Settings
@@ -101,6 +103,7 @@ class ChromaDocumentStore:
101
103
  self._host = host
102
104
  self._port = port
103
105
 
106
+ self._client: ClientAPI | None = None
104
107
  self._collection: chromadb.Collection | None = None
105
108
  self._async_collection: AsyncCollection | None = None
106
109
 
@@ -138,7 +141,7 @@ class ChromaDocumentStore:
138
141
  # Local persistent storage
139
142
  client = chromadb.PersistentClient(path=self._persist_path, **client_kwargs)
140
143
 
141
- self._client = client # store client for potential future use
144
+ self._client = client
142
145
 
143
146
  # Build the collection metadata locally so `self._metadata` stays exactly as the user passed it.
144
147
  # This keeps `to_dict()` deterministic and avoids mutating a user-supplied dict in place.
@@ -213,6 +216,15 @@ class ChromaDocumentStore:
213
216
  embedding_function=self._embedding_func,
214
217
  )
215
218
 
219
+ def close(self) -> None:
220
+ """Release the associated synchronous resources."""
221
+ if self._client is not None:
222
+ with suppress(Exception):
223
+ # `close` is not declared on the `ClientAPI` interface, but the concrete Chroma clients implement it
224
+ self._client.close() # type: ignore[attr-defined]
225
+ self._client = None
226
+ self._collection = None
227
+
216
228
  @staticmethod
217
229
  def _prepare_get_kwargs(filters: dict[str, Any] | None = None) -> dict[str, Any]:
218
230
  """
@@ -350,16 +362,17 @@ class ChromaDocumentStore:
350
362
  search_term: str | None,
351
363
  from_: int,
352
364
  size: int,
353
- ) -> tuple[list[str], int]:
365
+ ) -> tuple[list[Any], int]:
354
366
  """
355
367
  Computes paginated unique values for a metadata field from a Chroma get result.
356
368
 
357
369
  :param result: The full GetResult from a Chroma collection get operation.
358
370
  :param field_name: The metadata field name to collect unique values for.
359
- :param search_term: Optional search term to filter documents by content.
371
+ :param search_term: Optional search term to filter by, matched as a case-insensitive
372
+ substring against the value of `field_name`.
360
373
  :param from_: The offset to start returning values from.
361
374
  :param size: The maximum number of unique values to return.
362
- :returns: A tuple of the paginated unique values list and the total count.
375
+ :returns: A tuple of the paginated unique values (in their original type) list and the total count.
363
376
  """
364
377
  metadatas = result.get("metadatas", [])
365
378
 
@@ -367,24 +380,35 @@ class ChromaDocumentStore:
367
380
  return [], 0
368
381
 
369
382
  if search_term:
370
- documents = result.get("documents")
371
- if documents is None:
372
- documents = []
373
- filtered_metadatas = [
374
- metadatas[i] for i in range(len(documents)) if documents[i] and search_term in documents[i]
383
+ search_term_lower = search_term.lower()
384
+ metadatas = [
385
+ meta
386
+ for meta in metadatas
387
+ if meta
388
+ and field_name in meta
389
+ and meta.get(field_name) is not None
390
+ and search_term_lower in str(meta.get(field_name)).lower()
375
391
  ]
376
- metadatas = filtered_metadatas
377
392
 
378
393
  if not metadatas:
379
394
  return [], 0
380
395
 
381
- unique_values = {
382
- str(meta.get(field_name))
383
- for meta in metadatas
384
- if meta and field_name in meta and meta.get(field_name) is not None
385
- }
396
+ # Values of different types can be equal in Python (`1 == True`, `1 == 1.0`), so a plain set
397
+ # would silently merge them. Dedupe by (type, value) instead to keep them distinct.
398
+ seen: set[tuple[type, Any]] = set()
399
+ unique_values: list[Any] = []
400
+ for meta in metadatas:
401
+ if not meta or field_name not in meta:
402
+ continue
403
+ value = meta.get(field_name)
404
+ if value is None:
405
+ continue
406
+ dedup_key = (type(value), value)
407
+ if dedup_key not in seen:
408
+ seen.add(dedup_key)
409
+ unique_values.append(value)
386
410
 
387
- sorted_values = sorted(unique_values)
411
+ sorted_values = sorted(unique_values, key=str)
388
412
  total_count = len(sorted_values)
389
413
  end = from_ + size
390
414
  paginated_values = sorted_values[from_:end]
@@ -901,6 +925,7 @@ class ChromaDocumentStore:
901
925
  """
902
926
  self._ensure_initialized() # _ensure_initialized ensures _client is not None and a collection exists
903
927
  assert self._collection is not None
928
+ assert self._client is not None
904
929
 
905
930
  try:
906
931
  if recreate_index:
@@ -1292,27 +1317,27 @@ class ChromaDocumentStore:
1292
1317
  search_term: str | None = None,
1293
1318
  from_: int = 0,
1294
1319
  size: int = 10,
1295
- ) -> tuple[list[str], int]:
1320
+ filters: dict[str, Any] | None = None,
1321
+ ) -> tuple[list[Any], int]:
1296
1322
  """
1297
- Return unique metadata field values, optionally filtered by a content search term, with pagination.
1323
+ Return unique metadata field values, optionally filtered by a search term, with pagination.
1298
1324
 
1299
1325
  :param metadata_field: The metadata field to get unique values for.
1300
1326
  Can include or omit the "meta." prefix.
1301
- :param search_term: Optional search term to filter documents by matching
1302
- in the content field.
1327
+ :param search_term: Optional search term to filter values, matched as a
1328
+ case-insensitive substring against the metadata field's value.
1303
1329
  :param from_: The offset to start returning values from (for pagination).
1304
1330
  :param size: The maximum number of unique values to return.
1305
- :returns: A tuple containing list of unique values and total count of unique values.
1331
+ :param filters: Optional filters to restrict the documents considered.
1332
+ :returns: A tuple containing list of unique values (in their original type) and total count of unique values.
1306
1333
  """
1307
1334
  self._ensure_initialized()
1308
1335
  assert self._collection is not None
1309
1336
 
1310
1337
  field_name = _normalize_metadata_field_name(metadata_field)
1311
1338
 
1312
- kwargs: dict[str, Any] = {"include": ["metadatas"]}
1313
- if search_term:
1314
- kwargs["include"] = ["metadatas", "documents"]
1315
-
1339
+ kwargs = ChromaDocumentStore._prepare_get_kwargs(filters)
1340
+ kwargs["include"] = ["metadatas"]
1316
1341
  result = self._collection.get(**kwargs)
1317
1342
  return self._compute_field_unique_values(result, field_name, search_term, from_, size)
1318
1343
 
@@ -1322,29 +1347,29 @@ class ChromaDocumentStore:
1322
1347
  search_term: str | None = None,
1323
1348
  from_: int = 0,
1324
1349
  size: int = 10,
1325
- ) -> tuple[list[str], int]:
1350
+ filters: dict[str, Any] | None = None,
1351
+ ) -> tuple[list[Any], int]:
1326
1352
  """
1327
- Asynchronously return unique metadata field values, optionally filtered by content, with pagination.
1353
+ Asynchronously return unique metadata field values, optionally filtered by a search term, with pagination.
1328
1354
 
1329
1355
  Asynchronous methods are only supported for HTTP connections.
1330
1356
 
1331
1357
  :param metadata_field: The metadata field to get unique values for.
1332
1358
  Can include or omit the "meta." prefix.
1333
- :param search_term: Optional search term to filter documents by matching
1334
- in the content field.
1359
+ :param search_term: Optional search term to filter values, matched as a
1360
+ case-insensitive substring against the metadata field's value.
1335
1361
  :param from_: The offset to start returning values from (for pagination).
1336
1362
  :param size: The maximum number of unique values to return.
1337
- :returns: A tuple containing list of unique values and total count of unique values.
1363
+ :param filters: Optional filters to restrict the documents considered.
1364
+ :returns: A tuple containing list of unique values (in their original type) and total count of unique values.
1338
1365
  """
1339
1366
  await self._ensure_initialized_async()
1340
1367
  assert self._async_collection is not None
1341
1368
 
1342
1369
  field_name = _normalize_metadata_field_name(metadata_field)
1343
1370
 
1344
- kwargs: dict[str, Any] = {"include": ["metadatas"]}
1345
- if search_term:
1346
- kwargs["include"] = ["metadatas", "documents"]
1347
-
1371
+ kwargs = ChromaDocumentStore._prepare_get_kwargs(filters)
1372
+ kwargs["include"] = ["metadatas"]
1348
1373
  result = await self._async_collection.get(**kwargs)
1349
1374
  return self._compute_field_unique_values(result, field_name, search_term, from_, size)
1350
1375
 
@@ -180,6 +180,29 @@ class TestDocumentStoreUnit:
180
180
  with pytest.raises(ValueError, match="Invalid client_settings"):
181
181
  store._ensure_initialized()
182
182
 
183
+ def test_close(self):
184
+ store = ChromaDocumentStore()
185
+ client = mock.Mock()
186
+ store._client = client
187
+
188
+ store.close()
189
+
190
+ client.close.assert_called_once()
191
+ assert store._client is None
192
+
193
+ store.close()
194
+ client.close.assert_called_once()
195
+
196
+ def test_close_is_exception_safe(self):
197
+ store = ChromaDocumentStore()
198
+ client = mock.Mock()
199
+ client.close.side_effect = RuntimeError("boom")
200
+ store._client = client
201
+
202
+ store.close()
203
+
204
+ assert store._client is None
205
+
183
206
  def test_infer_type_from_value_fallback_for_unknown_type(self):
184
207
  assert ChromaDocumentStore._infer_type_from_value(None) == "keyword"
185
208
  assert ChromaDocumentStore._infer_type_from_value(["a", "b"]) == "keyword"
@@ -196,10 +219,11 @@ class TestDocumentStoreUnit:
196
219
  @pytest.mark.parametrize(
197
220
  "result",
198
221
  [
199
- {"ids": ["1"], "documents": None, "metadatas": [{"cat": "A"}]},
200
- {"ids": ["1"], "documents": ["hello world"], "metadatas": [{"cat": "A"}]},
222
+ {"ids": ["1"], "metadatas": [{"cat": "A"}]},
223
+ {"ids": ["1"], "metadatas": [None]},
224
+ {"ids": ["1"], "metadatas": [{"other": "A"}]},
201
225
  ],
202
- ids=["documents_none", "no_matches"],
226
+ ids=["no_match", "metadata_none", "field_missing"],
203
227
  )
204
228
  def test_compute_field_unique_values_with_search_term_edge_cases(self, result):
205
229
  values, total = ChromaDocumentStore._compute_field_unique_values(result, "cat", "absent", 0, 10)
@@ -377,6 +401,16 @@ class TestDocumentStore(
377
401
  assert store._collection.metadata["hnsw:space"] == "ip"
378
402
  assert new_store._collection.metadata["hnsw:space"] == "ip"
379
403
 
404
+ def test_close_and_reopen(self, tmp_path):
405
+ store = ChromaDocumentStore(collection_name="test_close_and_reopen", persist_path=str(tmp_path))
406
+ store.write_documents([Document(content="doc", embedding=TEST_EMBEDDING_1)])
407
+ assert store.count_documents() == 1
408
+
409
+ store.close()
410
+ assert store._client is None
411
+
412
+ assert store.count_documents() == 1
413
+
380
414
  def test_delete_empty(self, document_store: ChromaDocumentStore):
381
415
  """
382
416
  Deleting a non-existing document should not raise with Chroma
@@ -718,63 +752,6 @@ class TestMetadataOperations:
718
752
  assert min_max["min"] is None
719
753
  assert min_max["max"] is None
720
754
 
721
- def test_get_metadata_field_unique_values_basic(self, populated_store):
722
- """Test getting unique values for metadata field"""
723
- values, total = populated_store.get_metadata_field_unique_values("category", from_=0, size=10)
724
- assert sorted(values) == ["A", "B", "C"]
725
- assert total == 3
726
-
727
- def test_get_metadata_field_unique_values_pagination(self, populated_store):
728
- """Test pagination of unique values"""
729
- # First page
730
- values_page1, total = populated_store.get_metadata_field_unique_values("category", from_=0, size=2)
731
- assert len(values_page1) == 2
732
- assert total == 3
733
-
734
- # Second page
735
- values_page2, total = populated_store.get_metadata_field_unique_values("category", from_=2, size=2)
736
- assert len(values_page2) == 1
737
- assert total == 3
738
-
739
- # Check all values are returned across pages
740
- all_values = values_page1 + values_page2
741
- assert sorted(all_values) == ["A", "B", "C"]
742
-
743
- def test_get_metadata_field_unique_values_with_search_term(self, populated_store):
744
- """Test getting unique values filtered by search term"""
745
- # Search for documents containing "Doc 1"
746
- values, total = populated_store.get_metadata_field_unique_values(
747
- "category", search_term="Doc 1", from_=0, size=10
748
- )
749
- assert values == ["A"] # Only Doc 1 has category A
750
- assert total == 1
751
-
752
- def test_get_metadata_field_unique_values_field_normalization(self, populated_store):
753
- """Test field name normalization in unique values"""
754
- # Test with "meta." prefix
755
- values_with_prefix, total_with_prefix = populated_store.get_metadata_field_unique_values(
756
- "meta.category", from_=0, size=10
757
- )
758
- # Test without "meta." prefix
759
- values_without_prefix, total_without_prefix = populated_store.get_metadata_field_unique_values(
760
- "category", from_=0, size=10
761
- )
762
-
763
- assert sorted(values_with_prefix) == sorted(values_without_prefix) == ["A", "B", "C"]
764
- assert total_with_prefix == total_without_prefix == 3
765
-
766
- def test_get_metadata_field_unique_values_missing_field(self, populated_store):
767
- """Test getting unique values for non-existent field"""
768
- values, total = populated_store.get_metadata_field_unique_values("nonexistent_field", from_=0, size=10)
769
- assert values == []
770
- assert total == 0
771
-
772
- def test_get_metadata_field_unique_values_empty_collection(self, document_store):
773
- """Test getting unique values from empty collection"""
774
- values, total = document_store.get_metadata_field_unique_values("category", from_=0, size=10)
775
- assert values == []
776
- assert total == 0
777
-
778
755
  def test_get_metadata_field_unique_values_sorting(self, populated_store):
779
756
  """Test that unique values are sorted consistently"""
780
757
  values, total = populated_store.get_metadata_field_unique_values("status", from_=0, size=10)
@@ -123,7 +123,7 @@ class TestDocumentStoreAsync(
123
123
  assert store._async_client.get_settings().anonymized_telemetry is False
124
124
 
125
125
  async def test_search_async(self):
126
- document_store = ChromaDocumentStore(host="localhost", port=8000, collection_name="my_custom_collection")
126
+ document_store = ChromaDocumentStore(host="localhost", port=8000, collection_name=f"{uuid.uuid1()}-search")
127
127
 
128
128
  documents = [
129
129
  Document(content="First document", meta={"author": "Author1"}),
@@ -205,13 +205,3 @@ class TestDocumentStoreAsync(
205
205
  min_max = await document_store.get_metadata_field_min_max_async("nonexistent_field")
206
206
  assert min_max["min"] is None
207
207
  assert min_max["max"] is None
208
-
209
- async def test_get_metadata_field_unique_values_async_missing_field(self, document_store: ChromaDocumentStore):
210
- """Chroma-specific: unique values for non-existent field returns empty."""
211
- docs = [Document(content="Doc 1", meta={"category": "A"})]
212
- await document_store.write_documents_async(docs)
213
- values, total = await document_store.get_metadata_field_unique_values_async(
214
- "nonexistent_field", from_=0, size=10
215
- )
216
- assert values == []
217
- assert total == 0
@@ -106,6 +106,15 @@ class TestChromaQueryTextRetriever:
106
106
  assert retriever.top_k == 42
107
107
  assert retriever.filter_policy == FilterPolicy.REPLACE # default even if not specified
108
108
 
109
+ def test_close(self):
110
+ ds = mock.Mock(spec=ChromaDocumentStore)
111
+ retriever = ChromaQueryTextRetriever(ds)
112
+
113
+ retriever.close()
114
+
115
+ ds.close.assert_called_once()
116
+ assert retriever.document_store is ds
117
+
109
118
  def test_run_delegates_to_document_store_search(self):
110
119
  ds = mock.Mock(spec=ChromaDocumentStore)
111
120
  expected = [Document(content="hit")]
@@ -215,6 +224,15 @@ class TestChromaEmbeddingRetriever:
215
224
  retriever = ChromaEmbeddingRetriever.from_dict(data)
216
225
  assert retriever.filter_policy == FilterPolicy.REPLACE
217
226
 
227
+ def test_close(self):
228
+ ds = mock.Mock(spec=ChromaDocumentStore)
229
+ retriever = ChromaEmbeddingRetriever(ds)
230
+
231
+ retriever.close()
232
+
233
+ ds.close.assert_called_once()
234
+ assert retriever.document_store is ds
235
+
218
236
  def test_run_delegates_to_document_store_search_embeddings(self):
219
237
  ds = mock.Mock(spec=ChromaDocumentStore)
220
238
  expected = [Document(content="hit")]