crewai-memory-cosmosdb 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crewai_memory_cosmosdb-0.1.0/.gitignore +9 -0
- crewai_memory_cosmosdb-0.1.0/PKG-INFO +53 -0
- crewai_memory_cosmosdb-0.1.0/README.md +35 -0
- crewai_memory_cosmosdb-0.1.0/pyproject.toml +36 -0
- crewai_memory_cosmosdb-0.1.0/src/crewai_memory_cosmosdb/__init__.py +141 -0
- crewai_memory_cosmosdb-0.1.0/tests/test_cosmosdb_backend.py +46 -0
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: crewai-memory-cosmosdb
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Azure Cosmos DB StorageBackend for CrewAI unified Memory — hierarchical scopes, categories, metadata filters and native vector search (VectorDistance)
|
|
5
|
+
Project-URL: Homepage, https://github.com/skamalj/crewai-memory
|
|
6
|
+
Project-URL: Repository, https://github.com/skamalj/crewai-memory.git
|
|
7
|
+
Project-URL: Documentation, https://skamalj.github.io/agentstate-reducer/
|
|
8
|
+
Author-email: Kamal <skamalj@gmail.com>
|
|
9
|
+
Keywords: agent-memory,azure,cosmosdb,crewai,long-term-memory,memory,storage-backend,vector-search
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Requires-Dist: azure-cosmos>=4.7
|
|
16
|
+
Requires-Dist: crewai-memory-core>=0.1.0
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
|
|
19
|
+
# crewai-memory-cosmosdb
|
|
20
|
+
|
|
21
|
+
An **Azure Cosmos DB (NoSQL)** `StorageBackend` for [CrewAI](https://docs.crewai.com/en/concepts/memory)'s unified `Memory` — hierarchical scopes, categories, metadata filters, importance/recency, and **native vector search** via `VectorDistance` (diskANN index).
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install crewai-memory-cosmosdb
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
from crewai import Crew
|
|
29
|
+
from crewai.memory import Memory
|
|
30
|
+
from crewai_memory_cosmosdb import CosmosDBMemoryBackend
|
|
31
|
+
|
|
32
|
+
backend = CosmosDBMemoryBackend(
|
|
33
|
+
endpoint="https://<acct>.documents.azure.com:443/", key="<key>",
|
|
34
|
+
database_name="crewai", container_name="memory", dimensions=3072, # match your embedder
|
|
35
|
+
)
|
|
36
|
+
memory = Memory(storage=backend)
|
|
37
|
+
crew = Crew(agents=[...], tasks=[...], memory=memory)
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Or once for every `Crew(memory=True)`: `set_memory_storage_factory(lambda spec: CosmosDBMemoryBackend(...))`.
|
|
41
|
+
|
|
42
|
+
## How it works
|
|
43
|
+
|
|
44
|
+
- A container partitioned by `/scope`, one document per record (`id` = record id), created with a vector embedding policy on `/embedding` (cosine, `dimensions`) and a `diskANN` vector index (`vector_index_type="quantizedFlat"|"flat"` to change). The policy is fixed at container creation — use a new container to change `dimensions`.
|
|
45
|
+
- `search` is one SQL query: `ORDER BY VectorDistance(c.embedding, @q)` with the scope subtree in `WHERE`; the score is the cosine similarity. `categories` / `metadata_filter` / `min_score` are applied on the candidates.
|
|
46
|
+
- Scope tree, category counts, listing, `delete(older_than=...)`, `reset` are cross-partition SQL queries.
|
|
47
|
+
- The account needs **Vector Search for NoSQL**: `az cosmosdb update -n <acct> -g <rg> --capabilities EnableServerless EnableNoSQLVectorSearch` (list all existing capabilities; propagation can take a few minutes).
|
|
48
|
+
|
|
49
|
+
Docs: <https://skamalj.github.io/agentstate-reducer/> · part of [crewai-memory](https://github.com/skamalj/crewai-memory)
|
|
50
|
+
|
|
51
|
+
## License
|
|
52
|
+
|
|
53
|
+
MIT
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# crewai-memory-cosmosdb
|
|
2
|
+
|
|
3
|
+
An **Azure Cosmos DB (NoSQL)** `StorageBackend` for [CrewAI](https://docs.crewai.com/en/concepts/memory)'s unified `Memory` — hierarchical scopes, categories, metadata filters, importance/recency, and **native vector search** via `VectorDistance` (diskANN index).
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pip install crewai-memory-cosmosdb
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from crewai import Crew
|
|
11
|
+
from crewai.memory import Memory
|
|
12
|
+
from crewai_memory_cosmosdb import CosmosDBMemoryBackend
|
|
13
|
+
|
|
14
|
+
backend = CosmosDBMemoryBackend(
|
|
15
|
+
endpoint="https://<acct>.documents.azure.com:443/", key="<key>",
|
|
16
|
+
database_name="crewai", container_name="memory", dimensions=3072, # match your embedder
|
|
17
|
+
)
|
|
18
|
+
memory = Memory(storage=backend)
|
|
19
|
+
crew = Crew(agents=[...], tasks=[...], memory=memory)
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Or once for every `Crew(memory=True)`: `set_memory_storage_factory(lambda spec: CosmosDBMemoryBackend(...))`.
|
|
23
|
+
|
|
24
|
+
## How it works
|
|
25
|
+
|
|
26
|
+
- A container partitioned by `/scope`, one document per record (`id` = record id), created with a vector embedding policy on `/embedding` (cosine, `dimensions`) and a `diskANN` vector index (`vector_index_type="quantizedFlat"|"flat"` to change). The policy is fixed at container creation — use a new container to change `dimensions`.
|
|
27
|
+
- `search` is one SQL query: `ORDER BY VectorDistance(c.embedding, @q)` with the scope subtree in `WHERE`; the score is the cosine similarity. `categories` / `metadata_filter` / `min_score` are applied on the candidates.
|
|
28
|
+
- Scope tree, category counts, listing, `delete(older_than=...)`, `reset` are cross-partition SQL queries.
|
|
29
|
+
- The account needs **Vector Search for NoSQL**: `az cosmosdb update -n <acct> -g <rg> --capabilities EnableServerless EnableNoSQLVectorSearch` (list all existing capabilities; propagation can take a few minutes).
|
|
30
|
+
|
|
31
|
+
Docs: <https://skamalj.github.io/agentstate-reducer/> · part of [crewai-memory](https://github.com/skamalj/crewai-memory)
|
|
32
|
+
|
|
33
|
+
## License
|
|
34
|
+
|
|
35
|
+
MIT
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "crewai-memory-cosmosdb"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Azure Cosmos DB StorageBackend for CrewAI unified Memory — hierarchical scopes, categories, metadata filters and native vector search (VectorDistance)"
|
|
9
|
+
authors = [{name = "Kamal", email = "skamalj@gmail.com"}]
|
|
10
|
+
readme = "README.md"
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
dependencies = [
|
|
13
|
+
"crewai-memory-core>=0.1.0",
|
|
14
|
+
"azure-cosmos>=4.7",
|
|
15
|
+
]
|
|
16
|
+
keywords = ["crewai", "memory", "storage-backend", "cosmosdb", "azure", "long-term-memory", "agent-memory", "vector-search"]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"License :: OSI Approved :: MIT License",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
[project.urls]
|
|
25
|
+
Homepage = "https://github.com/skamalj/crewai-memory"
|
|
26
|
+
Repository = "https://github.com/skamalj/crewai-memory.git"
|
|
27
|
+
Documentation = "https://skamalj.github.io/agentstate-reducer/"
|
|
28
|
+
|
|
29
|
+
[dependency-groups]
|
|
30
|
+
dev = ["pytest>=7.0", "pytest-asyncio"]
|
|
31
|
+
|
|
32
|
+
[tool.pytest.ini_options]
|
|
33
|
+
asyncio_mode = "auto"
|
|
34
|
+
|
|
35
|
+
[tool.hatch.build.targets.wheel]
|
|
36
|
+
packages = ["src/crewai_memory_cosmosdb"]
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""Azure Cosmos DB (NoSQL) ``StorageBackend`` for CrewAI unified ``Memory``.
|
|
2
|
+
|
|
3
|
+
A container partitioned by ``/scope`` holding one document per ``MemoryRecord``
|
|
4
|
+
(``id`` = record id) with a **vector embedding policy** on ``/embedding``
|
|
5
|
+
(cosine) and a ``diskANN`` vector index. ``search`` is a single SQL query:
|
|
6
|
+
``ORDER BY VectorDistance(c.embedding, @q)`` with the scope subtree filtered in
|
|
7
|
+
``WHERE``. Built on ``crewai-memory-core``.
|
|
8
|
+
|
|
9
|
+
The account needs the ``EnableNoSQLVectorSearch`` capability; the vector policy
|
|
10
|
+
can only be set at container creation.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any, List, Optional, Tuple
|
|
16
|
+
|
|
17
|
+
from azure.cosmos import CosmosClient, PartitionKey
|
|
18
|
+
from azure.cosmos.exceptions import CosmosResourceNotFoundError
|
|
19
|
+
|
|
20
|
+
from crewai_memory_core import MemoryBackend, ROW_FIELDS, norm_scope
|
|
21
|
+
|
|
22
|
+
__all__ = ["CosmosDBMemoryBackend"]
|
|
23
|
+
|
|
24
|
+
_PROJ = ", ".join(f'c["{f}"]' for f in ROW_FIELDS if f != "embedding")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class CosmosDBMemoryBackend(MemoryBackend):
|
|
28
|
+
"""CrewAI ``StorageBackend`` on Azure Cosmos DB with native vector search.
|
|
29
|
+
|
|
30
|
+
Example:
|
|
31
|
+
```python
|
|
32
|
+
from crewai.memory import Memory
|
|
33
|
+
from crewai_memory_cosmosdb import CosmosDBMemoryBackend
|
|
34
|
+
|
|
35
|
+
backend = CosmosDBMemoryBackend(endpoint=..., key=..., database_name="crewai",
|
|
36
|
+
container_name="memory", dimensions=3072)
|
|
37
|
+
memory = Memory(storage=backend)
|
|
38
|
+
```
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
supports_native_vector_search = True
|
|
42
|
+
|
|
43
|
+
def __init__(
|
|
44
|
+
self,
|
|
45
|
+
*,
|
|
46
|
+
endpoint: str,
|
|
47
|
+
key: str,
|
|
48
|
+
database_name: str,
|
|
49
|
+
container_name: str,
|
|
50
|
+
dimensions: int = 3072,
|
|
51
|
+
vector_index_type: str = "diskANN",
|
|
52
|
+
) -> None:
|
|
53
|
+
self.dimensions = dimensions
|
|
54
|
+
client = CosmosClient(endpoint, credential=key)
|
|
55
|
+
db = client.create_database_if_not_exists(database_name)
|
|
56
|
+
self._container = db.create_container_if_not_exists(
|
|
57
|
+
id=container_name,
|
|
58
|
+
partition_key=PartitionKey(path="/scope"),
|
|
59
|
+
vector_embedding_policy={
|
|
60
|
+
"vectorEmbeddings": [
|
|
61
|
+
{"path": "/embedding", "dataType": "float32", "distanceFunction": "cosine", "dimensions": dimensions}
|
|
62
|
+
]
|
|
63
|
+
},
|
|
64
|
+
indexing_policy={
|
|
65
|
+
"automatic": True,
|
|
66
|
+
"indexingMode": "consistent",
|
|
67
|
+
"includedPaths": [{"path": "/*"}],
|
|
68
|
+
"excludedPaths": [{"path": "/embedding/*"}, {"path": '/"_etag"/?'}],
|
|
69
|
+
"vectorIndexes": [{"path": "/embedding", "type": vector_index_type}],
|
|
70
|
+
},
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
# ------------------------------------------------------------------ helpers
|
|
74
|
+
@staticmethod
|
|
75
|
+
def _row(doc: dict) -> dict:
|
|
76
|
+
row = {f: doc.get(f) for f in ROW_FIELDS}
|
|
77
|
+
row["scope"] = norm_scope(doc.get("scope"))
|
|
78
|
+
return row
|
|
79
|
+
|
|
80
|
+
@staticmethod
|
|
81
|
+
def _scope_where(scope_prefix: Optional[str]) -> Tuple[str, list]:
|
|
82
|
+
p = norm_scope(scope_prefix)
|
|
83
|
+
if p == "/":
|
|
84
|
+
return "true", []
|
|
85
|
+
return '(c["scope"] = @p OR STARTSWITH(c["scope"], @pc))', [
|
|
86
|
+
{"name": "@p", "value": p}, {"name": "@pc", "value": p + "/"},
|
|
87
|
+
]
|
|
88
|
+
|
|
89
|
+
def _query(self, sql: str, params: list) -> List[dict]:
|
|
90
|
+
return list(self._container.query_items(query=sql, parameters=params, enable_cross_partition_query=True))
|
|
91
|
+
|
|
92
|
+
# ------------------------------------------------------------------ primitives
|
|
93
|
+
def _put(self, rows: List[dict]) -> None:
|
|
94
|
+
for row in rows:
|
|
95
|
+
existing = self._get(row["id"])
|
|
96
|
+
if existing and existing["scope"] != norm_scope(row["scope"]):
|
|
97
|
+
self._delete_doc(row["id"], existing["scope"]) # scope moved: partition changes
|
|
98
|
+
doc = {f: row.get(f) for f in ROW_FIELDS}
|
|
99
|
+
doc["scope"] = norm_scope(row["scope"])
|
|
100
|
+
if not doc.get("embedding"):
|
|
101
|
+
doc.pop("embedding", None)
|
|
102
|
+
self._container.upsert_item(doc)
|
|
103
|
+
|
|
104
|
+
def _get(self, record_id: str) -> Optional[dict]:
|
|
105
|
+
docs = self._query('SELECT TOP 1 * FROM c WHERE c["id"] = @id', [{"name": "@id", "value": record_id}])
|
|
106
|
+
return self._row(docs[0]) if docs else None
|
|
107
|
+
|
|
108
|
+
def _delete_doc(self, record_id: str, scope: str) -> bool:
|
|
109
|
+
try:
|
|
110
|
+
self._container.delete_item(item=record_id, partition_key=scope)
|
|
111
|
+
return True
|
|
112
|
+
except CosmosResourceNotFoundError:
|
|
113
|
+
return False
|
|
114
|
+
|
|
115
|
+
def _delete_ids(self, ids: List[str]) -> int:
|
|
116
|
+
n = 0
|
|
117
|
+
for rid in ids:
|
|
118
|
+
row = self._get(rid)
|
|
119
|
+
if row and self._delete_doc(rid, row["scope"]):
|
|
120
|
+
n += 1
|
|
121
|
+
return n
|
|
122
|
+
|
|
123
|
+
def _scan(self, scope_prefix: Optional[str]) -> List[dict]:
|
|
124
|
+
where, params = self._scope_where(scope_prefix)
|
|
125
|
+
return [self._row(d) for d in self._query(f"SELECT * FROM c WHERE {where}", params)]
|
|
126
|
+
|
|
127
|
+
def _vector_search(
|
|
128
|
+
self, vector: List[float], scope_prefix: Optional[str], limit: int
|
|
129
|
+
) -> List[Tuple[dict, float]]:
|
|
130
|
+
where, params = self._scope_where(scope_prefix)
|
|
131
|
+
sql = (
|
|
132
|
+
f'SELECT TOP @k {_PROJ}, VectorDistance(c["embedding"], @q) AS score '
|
|
133
|
+
f'FROM c WHERE {where} AND IS_DEFINED(c["embedding"]) '
|
|
134
|
+
f'ORDER BY VectorDistance(c["embedding"], @q)'
|
|
135
|
+
)
|
|
136
|
+
params = params + [{"name": "@q", "value": vector}, {"name": "@k", "value": int(limit)}]
|
|
137
|
+
out: List[Tuple[dict, float]] = []
|
|
138
|
+
for d in self._query(sql, params):
|
|
139
|
+
row = self._row(d)
|
|
140
|
+
out.append((row, float(d["score"]))) # cosine similarity, higher is better
|
|
141
|
+
return out
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Cosmos DB StorageBackend: the shared contract suite against a real Cosmos account."""
|
|
2
|
+
import os
|
|
3
|
+
import uuid
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from crewai_memory_core.contract import * # noqa: F401,F403
|
|
8
|
+
from crewai_memory_core.contract import DIMS
|
|
9
|
+
from crewai_memory_cosmosdb import CosmosDBMemoryBackend
|
|
10
|
+
|
|
11
|
+
EP = os.environ.get("COSMOS_ENDPOINT")
|
|
12
|
+
KEY = os.environ.get("COSMOS_KEY")
|
|
13
|
+
DB = os.environ.get("CREWAI_COSMOS_DB", "crewai_memory_test")
|
|
14
|
+
|
|
15
|
+
pytestmark = pytest.mark.skipif(not (EP and KEY), reason="COSMOS_ENDPOINT/COSMOS_KEY not set")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@pytest.fixture(scope="module")
|
|
19
|
+
def _store():
|
|
20
|
+
cname = f"mem_{uuid.uuid4().hex[:8]}"
|
|
21
|
+
b = CosmosDBMemoryBackend(endpoint=EP, key=KEY, database_name=DB, container_name=cname, dimensions=DIMS)
|
|
22
|
+
yield b
|
|
23
|
+
from azure.cosmos import CosmosClient
|
|
24
|
+
CosmosClient(EP, credential=KEY).get_database_client(DB).delete_container(cname)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@pytest.fixture()
|
|
28
|
+
def backend(_store):
|
|
29
|
+
_store.reset()
|
|
30
|
+
yield _store
|
|
31
|
+
_store.reset()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_native_vector_path_used(backend, monkeypatch):
|
|
35
|
+
from crewai_memory_core.contract import _seed, EMB
|
|
36
|
+
_seed(backend)
|
|
37
|
+
seen = []
|
|
38
|
+
orig = backend._container.query_items
|
|
39
|
+
|
|
40
|
+
def spy(query, **kw):
|
|
41
|
+
seen.append(query)
|
|
42
|
+
return orig(query=query, **kw)
|
|
43
|
+
|
|
44
|
+
monkeypatch.setattr(backend._container, "query_items", spy)
|
|
45
|
+
assert backend.search(EMB(["sushi"])[0], scope_prefix="/crew", limit=5)
|
|
46
|
+
assert any("VectorDistance" in q for q in seen)
|