chroma-haystack 4.3.2__tar.gz → 4.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/CHANGELOG.md +18 -0
- chroma_haystack-4.3.2/LICENSE → chroma_haystack-4.5.0/LICENSE.txt +1 -1
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/PKG-INFO +5 -5
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/README.md +2 -2
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/pyproject.toml +1 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/components/retrievers/chroma/retriever.py +12 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/document_store.py +59 -34
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_document_store.py +37 -60
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_document_store_async.py +1 -11
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_retriever.py +18 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/.gitignore +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_01.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_02.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_03.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_04.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_05.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_06.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_07.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_08.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_09.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_10.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_11.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_12.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_20.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_21.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_22.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_23.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_24.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_25.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_26.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_27.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_28.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_29.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_30.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_31.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_32.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_40.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_41.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_42.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_43.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_44.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_45.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_46.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_50.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_51.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_52.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/data/usr_90.txt +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/example/example.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/pydoc/config_docusaurus.yml +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/components/retrievers/chroma/__init__.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/components/retrievers/py.typed +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/__init__.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/errors.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/filters.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/chroma/utils.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/py.typed +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/__init__.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/conftest.py +0 -0
- {chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/tests/test_filters.py +0 -0
|
@@ -1,5 +1,23 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [integrations/chroma-v4.4.0] - 2026-07-24
|
|
4
|
+
|
|
5
|
+
### 🚀 Features
|
|
6
|
+
|
|
7
|
+
- Chroma - add sync closing methods (#3660)
|
|
8
|
+
|
|
9
|
+
### 🐛 Bug Fixes
|
|
10
|
+
|
|
11
|
+
- `ChromaDocumentStore` match `search_term` against metadata field value, not content (#3644)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
## [integrations/chroma-v4.3.2] - 2026-07-17
|
|
15
|
+
|
|
16
|
+
### 🐛 Bug Fixes
|
|
17
|
+
|
|
18
|
+
- *(chroma)* Raise a clear error for id filters with a non-'==' operator (#3587)
|
|
19
|
+
|
|
20
|
+
|
|
3
21
|
## [integrations/chroma-v4.3.1] - 2026-06-29
|
|
4
22
|
|
|
5
23
|
### 🐛 Bug Fixes
|
|
@@ -186,7 +186,7 @@
|
|
|
186
186
|
same "printed page" as the copyright notice for easier
|
|
187
187
|
identification within third-party archives.
|
|
188
188
|
|
|
189
|
-
Copyright 2023 deepset GmbH
|
|
189
|
+
Copyright 2023-present deepset GmbH
|
|
190
190
|
|
|
191
191
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
192
192
|
you may not use this file except in compliance with the License.
|
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: chroma-haystack
|
|
3
|
-
Version: 4.
|
|
3
|
+
Version: 4.5.0
|
|
4
4
|
Project-URL: Documentation, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/chroma#readme
|
|
5
5
|
Project-URL: Issues, https://github.com/deepset-ai/haystack-core-integrations/issues
|
|
6
6
|
Project-URL: Source, https://github.com/deepset-ai/haystack-core-integrations/tree/main/integrations/chroma
|
|
7
7
|
Author-email: deepset GmbH <info@deepset.ai>
|
|
8
8
|
License-Expression: Apache-2.0
|
|
9
|
-
License-File: LICENSE
|
|
9
|
+
License-File: LICENSE.txt
|
|
10
10
|
Classifier: Development Status :: 4 - Beta
|
|
11
11
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
12
12
|
Classifier: Programming Language :: Python
|
|
@@ -36,5 +36,5 @@ Description-Content-Type: text/markdown
|
|
|
36
36
|
|
|
37
37
|
Refer to the general [Contribution Guidelines](https://github.com/deepset-ai/haystack-core-integrations/blob/main/CONTRIBUTING.md).
|
|
38
38
|
|
|
39
|
-
To run integration tests locally, you need a Chroma server running
|
|
40
|
-
|
|
39
|
+
To run integration tests locally, you need a Chroma server running on `localhost:8000`.
|
|
40
|
+
You can start it with: `hatch run chroma run`.
|
|
@@ -12,5 +12,5 @@
|
|
|
12
12
|
|
|
13
13
|
Refer to the general [Contribution Guidelines](https://github.com/deepset-ai/haystack-core-integrations/blob/main/CONTRIBUTING.md).
|
|
14
14
|
|
|
15
|
-
To run integration tests locally, you need a Chroma server running
|
|
16
|
-
|
|
15
|
+
To run integration tests locally, you need a Chroma server running on `localhost:8000`.
|
|
16
|
+
You can start it with: `hatch run chroma run`.
|
|
@@ -119,6 +119,12 @@ class ChromaQueryTextRetriever:
|
|
|
119
119
|
top_k = top_k or self.top_k
|
|
120
120
|
return {"documents": (await self.document_store.search_async([query], top_k, filters))[0]}
|
|
121
121
|
|
|
122
|
+
def close(self) -> None:
|
|
123
|
+
"""
|
|
124
|
+
Release the synchronous resources of the underlying Document Store.
|
|
125
|
+
"""
|
|
126
|
+
self.document_store.close()
|
|
127
|
+
|
|
122
128
|
@classmethod
|
|
123
129
|
def from_dict(cls, data: dict[str, Any]) -> "ChromaQueryTextRetriever":
|
|
124
130
|
"""
|
|
@@ -238,6 +244,12 @@ class ChromaEmbeddingRetriever:
|
|
|
238
244
|
query_embeddings = [query_embedding]
|
|
239
245
|
return {"documents": (await self.document_store.search_embeddings_async(query_embeddings, top_k, filters))[0]}
|
|
240
246
|
|
|
247
|
+
def close(self) -> None:
|
|
248
|
+
"""
|
|
249
|
+
Release the synchronous resources of the underlying Document Store.
|
|
250
|
+
"""
|
|
251
|
+
self.document_store.close()
|
|
252
|
+
|
|
241
253
|
@classmethod
|
|
242
254
|
def from_dict(cls, data: dict[str, Any]) -> "ChromaEmbeddingRetriever":
|
|
243
255
|
"""
|
|
@@ -3,9 +3,11 @@
|
|
|
3
3
|
# SPDX-License-Identifier: Apache-2.0
|
|
4
4
|
|
|
5
5
|
from collections.abc import Sequence
|
|
6
|
+
from contextlib import suppress
|
|
6
7
|
from typing import Any, Literal, cast
|
|
7
8
|
|
|
8
9
|
import chromadb
|
|
10
|
+
from chromadb.api import ClientAPI
|
|
9
11
|
from chromadb.api.models.AsyncCollection import AsyncCollection
|
|
10
12
|
from chromadb.api.types import GetResult, Metadata, OneOrMany, QueryResult
|
|
11
13
|
from chromadb.config import Settings
|
|
@@ -101,6 +103,7 @@ class ChromaDocumentStore:
|
|
|
101
103
|
self._host = host
|
|
102
104
|
self._port = port
|
|
103
105
|
|
|
106
|
+
self._client: ClientAPI | None = None
|
|
104
107
|
self._collection: chromadb.Collection | None = None
|
|
105
108
|
self._async_collection: AsyncCollection | None = None
|
|
106
109
|
|
|
@@ -138,7 +141,7 @@ class ChromaDocumentStore:
|
|
|
138
141
|
# Local persistent storage
|
|
139
142
|
client = chromadb.PersistentClient(path=self._persist_path, **client_kwargs)
|
|
140
143
|
|
|
141
|
-
self._client = client
|
|
144
|
+
self._client = client
|
|
142
145
|
|
|
143
146
|
# Build the collection metadata locally so `self._metadata` stays exactly as the user passed it.
|
|
144
147
|
# This keeps `to_dict()` deterministic and avoids mutating a user-supplied dict in place.
|
|
@@ -213,6 +216,15 @@ class ChromaDocumentStore:
|
|
|
213
216
|
embedding_function=self._embedding_func,
|
|
214
217
|
)
|
|
215
218
|
|
|
219
|
+
def close(self) -> None:
|
|
220
|
+
"""Release the associated synchronous resources."""
|
|
221
|
+
if self._client is not None:
|
|
222
|
+
with suppress(Exception):
|
|
223
|
+
# `close` is not declared on the `ClientAPI` interface, but the concrete Chroma clients implement it
|
|
224
|
+
self._client.close() # type: ignore[attr-defined]
|
|
225
|
+
self._client = None
|
|
226
|
+
self._collection = None
|
|
227
|
+
|
|
216
228
|
@staticmethod
|
|
217
229
|
def _prepare_get_kwargs(filters: dict[str, Any] | None = None) -> dict[str, Any]:
|
|
218
230
|
"""
|
|
@@ -350,16 +362,17 @@ class ChromaDocumentStore:
|
|
|
350
362
|
search_term: str | None,
|
|
351
363
|
from_: int,
|
|
352
364
|
size: int,
|
|
353
|
-
) -> tuple[list[
|
|
365
|
+
) -> tuple[list[Any], int]:
|
|
354
366
|
"""
|
|
355
367
|
Computes paginated unique values for a metadata field from a Chroma get result.
|
|
356
368
|
|
|
357
369
|
:param result: The full GetResult from a Chroma collection get operation.
|
|
358
370
|
:param field_name: The metadata field name to collect unique values for.
|
|
359
|
-
:param search_term: Optional search term to filter
|
|
371
|
+
:param search_term: Optional search term to filter by, matched as a case-insensitive
|
|
372
|
+
substring against the value of `field_name`.
|
|
360
373
|
:param from_: The offset to start returning values from.
|
|
361
374
|
:param size: The maximum number of unique values to return.
|
|
362
|
-
:returns: A tuple of the paginated unique values list and the total count.
|
|
375
|
+
:returns: A tuple of the paginated unique values (in their original type) list and the total count.
|
|
363
376
|
"""
|
|
364
377
|
metadatas = result.get("metadatas", [])
|
|
365
378
|
|
|
@@ -367,24 +380,35 @@ class ChromaDocumentStore:
|
|
|
367
380
|
return [], 0
|
|
368
381
|
|
|
369
382
|
if search_term:
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
383
|
+
search_term_lower = search_term.lower()
|
|
384
|
+
metadatas = [
|
|
385
|
+
meta
|
|
386
|
+
for meta in metadatas
|
|
387
|
+
if meta
|
|
388
|
+
and field_name in meta
|
|
389
|
+
and meta.get(field_name) is not None
|
|
390
|
+
and search_term_lower in str(meta.get(field_name)).lower()
|
|
375
391
|
]
|
|
376
|
-
metadatas = filtered_metadatas
|
|
377
392
|
|
|
378
393
|
if not metadatas:
|
|
379
394
|
return [], 0
|
|
380
395
|
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
396
|
+
# Values of different types can be equal in Python (`1 == True`, `1 == 1.0`), so a plain set
|
|
397
|
+
# would silently merge them. Dedupe by (type, value) instead to keep them distinct.
|
|
398
|
+
seen: set[tuple[type, Any]] = set()
|
|
399
|
+
unique_values: list[Any] = []
|
|
400
|
+
for meta in metadatas:
|
|
401
|
+
if not meta or field_name not in meta:
|
|
402
|
+
continue
|
|
403
|
+
value = meta.get(field_name)
|
|
404
|
+
if value is None:
|
|
405
|
+
continue
|
|
406
|
+
dedup_key = (type(value), value)
|
|
407
|
+
if dedup_key not in seen:
|
|
408
|
+
seen.add(dedup_key)
|
|
409
|
+
unique_values.append(value)
|
|
386
410
|
|
|
387
|
-
sorted_values = sorted(unique_values)
|
|
411
|
+
sorted_values = sorted(unique_values, key=str)
|
|
388
412
|
total_count = len(sorted_values)
|
|
389
413
|
end = from_ + size
|
|
390
414
|
paginated_values = sorted_values[from_:end]
|
|
@@ -901,6 +925,7 @@ class ChromaDocumentStore:
|
|
|
901
925
|
"""
|
|
902
926
|
self._ensure_initialized() # _ensure_initialized ensures _client is not None and a collection exists
|
|
903
927
|
assert self._collection is not None
|
|
928
|
+
assert self._client is not None
|
|
904
929
|
|
|
905
930
|
try:
|
|
906
931
|
if recreate_index:
|
|
@@ -1292,27 +1317,27 @@ class ChromaDocumentStore:
|
|
|
1292
1317
|
search_term: str | None = None,
|
|
1293
1318
|
from_: int = 0,
|
|
1294
1319
|
size: int = 10,
|
|
1295
|
-
|
|
1320
|
+
filters: dict[str, Any] | None = None,
|
|
1321
|
+
) -> tuple[list[Any], int]:
|
|
1296
1322
|
"""
|
|
1297
|
-
Return unique metadata field values, optionally filtered by a
|
|
1323
|
+
Return unique metadata field values, optionally filtered by a search term, with pagination.
|
|
1298
1324
|
|
|
1299
1325
|
:param metadata_field: The metadata field to get unique values for.
|
|
1300
1326
|
Can include or omit the "meta." prefix.
|
|
1301
|
-
:param search_term: Optional search term to filter
|
|
1302
|
-
|
|
1327
|
+
:param search_term: Optional search term to filter values, matched as a
|
|
1328
|
+
case-insensitive substring against the metadata field's value.
|
|
1303
1329
|
:param from_: The offset to start returning values from (for pagination).
|
|
1304
1330
|
:param size: The maximum number of unique values to return.
|
|
1305
|
-
:
|
|
1331
|
+
:param filters: Optional filters to restrict the documents considered.
|
|
1332
|
+
:returns: A tuple containing list of unique values (in their original type) and total count of unique values.
|
|
1306
1333
|
"""
|
|
1307
1334
|
self._ensure_initialized()
|
|
1308
1335
|
assert self._collection is not None
|
|
1309
1336
|
|
|
1310
1337
|
field_name = _normalize_metadata_field_name(metadata_field)
|
|
1311
1338
|
|
|
1312
|
-
kwargs
|
|
1313
|
-
|
|
1314
|
-
kwargs["include"] = ["metadatas", "documents"]
|
|
1315
|
-
|
|
1339
|
+
kwargs = ChromaDocumentStore._prepare_get_kwargs(filters)
|
|
1340
|
+
kwargs["include"] = ["metadatas"]
|
|
1316
1341
|
result = self._collection.get(**kwargs)
|
|
1317
1342
|
return self._compute_field_unique_values(result, field_name, search_term, from_, size)
|
|
1318
1343
|
|
|
@@ -1322,29 +1347,29 @@ class ChromaDocumentStore:
|
|
|
1322
1347
|
search_term: str | None = None,
|
|
1323
1348
|
from_: int = 0,
|
|
1324
1349
|
size: int = 10,
|
|
1325
|
-
|
|
1350
|
+
filters: dict[str, Any] | None = None,
|
|
1351
|
+
) -> tuple[list[Any], int]:
|
|
1326
1352
|
"""
|
|
1327
|
-
Asynchronously return unique metadata field values, optionally filtered by
|
|
1353
|
+
Asynchronously return unique metadata field values, optionally filtered by a search term, with pagination.
|
|
1328
1354
|
|
|
1329
1355
|
Asynchronous methods are only supported for HTTP connections.
|
|
1330
1356
|
|
|
1331
1357
|
:param metadata_field: The metadata field to get unique values for.
|
|
1332
1358
|
Can include or omit the "meta." prefix.
|
|
1333
|
-
:param search_term: Optional search term to filter
|
|
1334
|
-
|
|
1359
|
+
:param search_term: Optional search term to filter values, matched as a
|
|
1360
|
+
case-insensitive substring against the metadata field's value.
|
|
1335
1361
|
:param from_: The offset to start returning values from (for pagination).
|
|
1336
1362
|
:param size: The maximum number of unique values to return.
|
|
1337
|
-
:
|
|
1363
|
+
:param filters: Optional filters to restrict the documents considered.
|
|
1364
|
+
:returns: A tuple containing list of unique values (in their original type) and total count of unique values.
|
|
1338
1365
|
"""
|
|
1339
1366
|
await self._ensure_initialized_async()
|
|
1340
1367
|
assert self._async_collection is not None
|
|
1341
1368
|
|
|
1342
1369
|
field_name = _normalize_metadata_field_name(metadata_field)
|
|
1343
1370
|
|
|
1344
|
-
kwargs
|
|
1345
|
-
|
|
1346
|
-
kwargs["include"] = ["metadatas", "documents"]
|
|
1347
|
-
|
|
1371
|
+
kwargs = ChromaDocumentStore._prepare_get_kwargs(filters)
|
|
1372
|
+
kwargs["include"] = ["metadatas"]
|
|
1348
1373
|
result = await self._async_collection.get(**kwargs)
|
|
1349
1374
|
return self._compute_field_unique_values(result, field_name, search_term, from_, size)
|
|
1350
1375
|
|
|
@@ -180,6 +180,29 @@ class TestDocumentStoreUnit:
|
|
|
180
180
|
with pytest.raises(ValueError, match="Invalid client_settings"):
|
|
181
181
|
store._ensure_initialized()
|
|
182
182
|
|
|
183
|
+
def test_close(self):
|
|
184
|
+
store = ChromaDocumentStore()
|
|
185
|
+
client = mock.Mock()
|
|
186
|
+
store._client = client
|
|
187
|
+
|
|
188
|
+
store.close()
|
|
189
|
+
|
|
190
|
+
client.close.assert_called_once()
|
|
191
|
+
assert store._client is None
|
|
192
|
+
|
|
193
|
+
store.close()
|
|
194
|
+
client.close.assert_called_once()
|
|
195
|
+
|
|
196
|
+
def test_close_is_exception_safe(self):
|
|
197
|
+
store = ChromaDocumentStore()
|
|
198
|
+
client = mock.Mock()
|
|
199
|
+
client.close.side_effect = RuntimeError("boom")
|
|
200
|
+
store._client = client
|
|
201
|
+
|
|
202
|
+
store.close()
|
|
203
|
+
|
|
204
|
+
assert store._client is None
|
|
205
|
+
|
|
183
206
|
def test_infer_type_from_value_fallback_for_unknown_type(self):
|
|
184
207
|
assert ChromaDocumentStore._infer_type_from_value(None) == "keyword"
|
|
185
208
|
assert ChromaDocumentStore._infer_type_from_value(["a", "b"]) == "keyword"
|
|
@@ -196,10 +219,11 @@ class TestDocumentStoreUnit:
|
|
|
196
219
|
@pytest.mark.parametrize(
|
|
197
220
|
"result",
|
|
198
221
|
[
|
|
199
|
-
{"ids": ["1"], "
|
|
200
|
-
{"ids": ["1"], "
|
|
222
|
+
{"ids": ["1"], "metadatas": [{"cat": "A"}]},
|
|
223
|
+
{"ids": ["1"], "metadatas": [None]},
|
|
224
|
+
{"ids": ["1"], "metadatas": [{"other": "A"}]},
|
|
201
225
|
],
|
|
202
|
-
ids=["
|
|
226
|
+
ids=["no_match", "metadata_none", "field_missing"],
|
|
203
227
|
)
|
|
204
228
|
def test_compute_field_unique_values_with_search_term_edge_cases(self, result):
|
|
205
229
|
values, total = ChromaDocumentStore._compute_field_unique_values(result, "cat", "absent", 0, 10)
|
|
@@ -377,6 +401,16 @@ class TestDocumentStore(
|
|
|
377
401
|
assert store._collection.metadata["hnsw:space"] == "ip"
|
|
378
402
|
assert new_store._collection.metadata["hnsw:space"] == "ip"
|
|
379
403
|
|
|
404
|
+
def test_close_and_reopen(self, tmp_path):
|
|
405
|
+
store = ChromaDocumentStore(collection_name="test_close_and_reopen", persist_path=str(tmp_path))
|
|
406
|
+
store.write_documents([Document(content="doc", embedding=TEST_EMBEDDING_1)])
|
|
407
|
+
assert store.count_documents() == 1
|
|
408
|
+
|
|
409
|
+
store.close()
|
|
410
|
+
assert store._client is None
|
|
411
|
+
|
|
412
|
+
assert store.count_documents() == 1
|
|
413
|
+
|
|
380
414
|
def test_delete_empty(self, document_store: ChromaDocumentStore):
|
|
381
415
|
"""
|
|
382
416
|
Deleting a non-existing document should not raise with Chroma
|
|
@@ -718,63 +752,6 @@ class TestMetadataOperations:
|
|
|
718
752
|
assert min_max["min"] is None
|
|
719
753
|
assert min_max["max"] is None
|
|
720
754
|
|
|
721
|
-
def test_get_metadata_field_unique_values_basic(self, populated_store):
|
|
722
|
-
"""Test getting unique values for metadata field"""
|
|
723
|
-
values, total = populated_store.get_metadata_field_unique_values("category", from_=0, size=10)
|
|
724
|
-
assert sorted(values) == ["A", "B", "C"]
|
|
725
|
-
assert total == 3
|
|
726
|
-
|
|
727
|
-
def test_get_metadata_field_unique_values_pagination(self, populated_store):
|
|
728
|
-
"""Test pagination of unique values"""
|
|
729
|
-
# First page
|
|
730
|
-
values_page1, total = populated_store.get_metadata_field_unique_values("category", from_=0, size=2)
|
|
731
|
-
assert len(values_page1) == 2
|
|
732
|
-
assert total == 3
|
|
733
|
-
|
|
734
|
-
# Second page
|
|
735
|
-
values_page2, total = populated_store.get_metadata_field_unique_values("category", from_=2, size=2)
|
|
736
|
-
assert len(values_page2) == 1
|
|
737
|
-
assert total == 3
|
|
738
|
-
|
|
739
|
-
# Check all values are returned across pages
|
|
740
|
-
all_values = values_page1 + values_page2
|
|
741
|
-
assert sorted(all_values) == ["A", "B", "C"]
|
|
742
|
-
|
|
743
|
-
def test_get_metadata_field_unique_values_with_search_term(self, populated_store):
|
|
744
|
-
"""Test getting unique values filtered by search term"""
|
|
745
|
-
# Search for documents containing "Doc 1"
|
|
746
|
-
values, total = populated_store.get_metadata_field_unique_values(
|
|
747
|
-
"category", search_term="Doc 1", from_=0, size=10
|
|
748
|
-
)
|
|
749
|
-
assert values == ["A"] # Only Doc 1 has category A
|
|
750
|
-
assert total == 1
|
|
751
|
-
|
|
752
|
-
def test_get_metadata_field_unique_values_field_normalization(self, populated_store):
|
|
753
|
-
"""Test field name normalization in unique values"""
|
|
754
|
-
# Test with "meta." prefix
|
|
755
|
-
values_with_prefix, total_with_prefix = populated_store.get_metadata_field_unique_values(
|
|
756
|
-
"meta.category", from_=0, size=10
|
|
757
|
-
)
|
|
758
|
-
# Test without "meta." prefix
|
|
759
|
-
values_without_prefix, total_without_prefix = populated_store.get_metadata_field_unique_values(
|
|
760
|
-
"category", from_=0, size=10
|
|
761
|
-
)
|
|
762
|
-
|
|
763
|
-
assert sorted(values_with_prefix) == sorted(values_without_prefix) == ["A", "B", "C"]
|
|
764
|
-
assert total_with_prefix == total_without_prefix == 3
|
|
765
|
-
|
|
766
|
-
def test_get_metadata_field_unique_values_missing_field(self, populated_store):
|
|
767
|
-
"""Test getting unique values for non-existent field"""
|
|
768
|
-
values, total = populated_store.get_metadata_field_unique_values("nonexistent_field", from_=0, size=10)
|
|
769
|
-
assert values == []
|
|
770
|
-
assert total == 0
|
|
771
|
-
|
|
772
|
-
def test_get_metadata_field_unique_values_empty_collection(self, document_store):
|
|
773
|
-
"""Test getting unique values from empty collection"""
|
|
774
|
-
values, total = document_store.get_metadata_field_unique_values("category", from_=0, size=10)
|
|
775
|
-
assert values == []
|
|
776
|
-
assert total == 0
|
|
777
|
-
|
|
778
755
|
def test_get_metadata_field_unique_values_sorting(self, populated_store):
|
|
779
756
|
"""Test that unique values are sorted consistently"""
|
|
780
757
|
values, total = populated_store.get_metadata_field_unique_values("status", from_=0, size=10)
|
|
@@ -123,7 +123,7 @@ class TestDocumentStoreAsync(
|
|
|
123
123
|
assert store._async_client.get_settings().anonymized_telemetry is False
|
|
124
124
|
|
|
125
125
|
async def test_search_async(self):
|
|
126
|
-
document_store = ChromaDocumentStore(host="localhost", port=8000, collection_name="
|
|
126
|
+
document_store = ChromaDocumentStore(host="localhost", port=8000, collection_name=f"{uuid.uuid1()}-search")
|
|
127
127
|
|
|
128
128
|
documents = [
|
|
129
129
|
Document(content="First document", meta={"author": "Author1"}),
|
|
@@ -205,13 +205,3 @@ class TestDocumentStoreAsync(
|
|
|
205
205
|
min_max = await document_store.get_metadata_field_min_max_async("nonexistent_field")
|
|
206
206
|
assert min_max["min"] is None
|
|
207
207
|
assert min_max["max"] is None
|
|
208
|
-
|
|
209
|
-
async def test_get_metadata_field_unique_values_async_missing_field(self, document_store: ChromaDocumentStore):
|
|
210
|
-
"""Chroma-specific: unique values for non-existent field returns empty."""
|
|
211
|
-
docs = [Document(content="Doc 1", meta={"category": "A"})]
|
|
212
|
-
await document_store.write_documents_async(docs)
|
|
213
|
-
values, total = await document_store.get_metadata_field_unique_values_async(
|
|
214
|
-
"nonexistent_field", from_=0, size=10
|
|
215
|
-
)
|
|
216
|
-
assert values == []
|
|
217
|
-
assert total == 0
|
|
@@ -106,6 +106,15 @@ class TestChromaQueryTextRetriever:
|
|
|
106
106
|
assert retriever.top_k == 42
|
|
107
107
|
assert retriever.filter_policy == FilterPolicy.REPLACE # default even if not specified
|
|
108
108
|
|
|
109
|
+
def test_close(self):
|
|
110
|
+
ds = mock.Mock(spec=ChromaDocumentStore)
|
|
111
|
+
retriever = ChromaQueryTextRetriever(ds)
|
|
112
|
+
|
|
113
|
+
retriever.close()
|
|
114
|
+
|
|
115
|
+
ds.close.assert_called_once()
|
|
116
|
+
assert retriever.document_store is ds
|
|
117
|
+
|
|
109
118
|
def test_run_delegates_to_document_store_search(self):
|
|
110
119
|
ds = mock.Mock(spec=ChromaDocumentStore)
|
|
111
120
|
expected = [Document(content="hit")]
|
|
@@ -215,6 +224,15 @@ class TestChromaEmbeddingRetriever:
|
|
|
215
224
|
retriever = ChromaEmbeddingRetriever.from_dict(data)
|
|
216
225
|
assert retriever.filter_policy == FilterPolicy.REPLACE
|
|
217
226
|
|
|
227
|
+
def test_close(self):
|
|
228
|
+
ds = mock.Mock(spec=ChromaDocumentStore)
|
|
229
|
+
retriever = ChromaEmbeddingRetriever(ds)
|
|
230
|
+
|
|
231
|
+
retriever.close()
|
|
232
|
+
|
|
233
|
+
ds.close.assert_called_once()
|
|
234
|
+
assert retriever.document_store is ds
|
|
235
|
+
|
|
218
236
|
def test_run_delegates_to_document_store_search_embeddings(self):
|
|
219
237
|
ds = mock.Mock(spec=ChromaDocumentStore)
|
|
220
238
|
expected = [Document(content="hit")]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{chroma_haystack-4.3.2 → chroma_haystack-4.5.0}/src/haystack_integrations/document_stores/py.typed
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|