crewai-memory-cosmosdb 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,9 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .pytest_cache/
5
+ .venv/
6
+ dist/
7
+ build/
8
+ uv.lock
9
+ .env
@@ -0,0 +1,53 @@
1
+ Metadata-Version: 2.5
2
+ Name: crewai-memory-cosmosdb
3
+ Version: 0.1.0
4
+ Summary: Azure Cosmos DB StorageBackend for CrewAI unified Memory — hierarchical scopes, categories, metadata filters and native vector search (VectorDistance)
5
+ Project-URL: Homepage, https://github.com/skamalj/crewai-memory
6
+ Project-URL: Repository, https://github.com/skamalj/crewai-memory.git
7
+ Project-URL: Documentation, https://skamalj.github.io/agentstate-reducer/
8
+ Author-email: Kamal <skamalj@gmail.com>
9
+ Keywords: agent-memory,azure,cosmosdb,crewai,long-term-memory,memory,storage-backend,vector-search
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
14
+ Requires-Python: >=3.10
15
+ Requires-Dist: azure-cosmos>=4.7
16
+ Requires-Dist: crewai-memory-core>=0.1.0
17
+ Description-Content-Type: text/markdown
18
+
19
+ # crewai-memory-cosmosdb
20
+
21
+ An **Azure Cosmos DB (NoSQL)** `StorageBackend` for [CrewAI](https://docs.crewai.com/en/concepts/memory)'s unified `Memory` — hierarchical scopes, categories, metadata filters, importance/recency, and **native vector search** via `VectorDistance` (diskANN index).
22
+
23
+ ```bash
24
+ pip install crewai-memory-cosmosdb
25
+ ```
26
+
27
+ ```python
28
+ from crewai import Crew
29
+ from crewai.memory import Memory
30
+ from crewai_memory_cosmosdb import CosmosDBMemoryBackend
31
+
32
+ backend = CosmosDBMemoryBackend(
33
+ endpoint="https://<acct>.documents.azure.com:443/", key="<key>",
34
+ database_name="crewai", container_name="memory", dimensions=3072, # match your embedder
35
+ )
36
+ memory = Memory(storage=backend)
37
+ crew = Crew(agents=[...], tasks=[...], memory=memory)
38
+ ```
39
+
40
+ Or once for every `Crew(memory=True)`: `set_memory_storage_factory(lambda spec: CosmosDBMemoryBackend(...))`.
41
+
42
+ ## How it works
43
+
44
+ - A container partitioned by `/scope`, one document per record (`id` = record id), created with a vector embedding policy on `/embedding` (cosine, `dimensions`) and a `diskANN` vector index (`vector_index_type="quantizedFlat"|"flat"` to change). The policy is fixed at container creation — use a new container to change `dimensions`.
45
+ - `search` is one SQL query: `ORDER BY VectorDistance(c.embedding, @q)` with the scope subtree in `WHERE`; the score is the cosine similarity. `categories` / `metadata_filter` / `min_score` are applied on the candidates.
46
+ - Scope tree, category counts, listing, `delete(older_than=...)`, `reset` are cross-partition SQL queries.
47
+ - The account needs **Vector Search for NoSQL**: `az cosmosdb update -n <acct> -g <rg> --capabilities EnableServerless EnableNoSQLVectorSearch` (list all existing capabilities; propagation can take a few minutes).
48
+
49
+ Docs: <https://skamalj.github.io/agentstate-reducer/> · part of [crewai-memory](https://github.com/skamalj/crewai-memory)
50
+
51
+ ## License
52
+
53
+ MIT
@@ -0,0 +1,35 @@
1
+ # crewai-memory-cosmosdb
2
+
3
+ An **Azure Cosmos DB (NoSQL)** `StorageBackend` for [CrewAI](https://docs.crewai.com/en/concepts/memory)'s unified `Memory` — hierarchical scopes, categories, metadata filters, importance/recency, and **native vector search** via `VectorDistance` (diskANN index).
4
+
5
+ ```bash
6
+ pip install crewai-memory-cosmosdb
7
+ ```
8
+
9
+ ```python
10
+ from crewai import Crew
11
+ from crewai.memory import Memory
12
+ from crewai_memory_cosmosdb import CosmosDBMemoryBackend
13
+
14
+ backend = CosmosDBMemoryBackend(
15
+ endpoint="https://<acct>.documents.azure.com:443/", key="<key>",
16
+ database_name="crewai", container_name="memory", dimensions=3072, # match your embedder
17
+ )
18
+ memory = Memory(storage=backend)
19
+ crew = Crew(agents=[...], tasks=[...], memory=memory)
20
+ ```
21
+
22
+ Or once for every `Crew(memory=True)`: `set_memory_storage_factory(lambda spec: CosmosDBMemoryBackend(...))`.
23
+
24
+ ## How it works
25
+
26
+ - A container partitioned by `/scope`, one document per record (`id` = record id), created with a vector embedding policy on `/embedding` (cosine, `dimensions`) and a `diskANN` vector index (`vector_index_type="quantizedFlat"|"flat"` to change). The policy is fixed at container creation — use a new container to change `dimensions`.
27
+ - `search` is one SQL query: `ORDER BY VectorDistance(c.embedding, @q)` with the scope subtree in `WHERE`; the score is the cosine similarity. `categories` / `metadata_filter` / `min_score` are applied on the candidates.
28
+ - Scope tree, category counts, listing, `delete(older_than=...)`, `reset` are cross-partition SQL queries.
29
+ - The account needs **Vector Search for NoSQL**: `az cosmosdb update -n <acct> -g <rg> --capabilities EnableServerless EnableNoSQLVectorSearch` (list all existing capabilities; propagation can take a few minutes).
30
+
31
+ Docs: <https://skamalj.github.io/agentstate-reducer/> · part of [crewai-memory](https://github.com/skamalj/crewai-memory)
32
+
33
+ ## License
34
+
35
+ MIT
@@ -0,0 +1,36 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "crewai-memory-cosmosdb"
7
+ version = "0.1.0"
8
+ description = "Azure Cosmos DB StorageBackend for CrewAI unified Memory — hierarchical scopes, categories, metadata filters and native vector search (VectorDistance)"
9
+ authors = [{name = "Kamal", email = "skamalj@gmail.com"}]
10
+ readme = "README.md"
11
+ requires-python = ">=3.10"
12
+ dependencies = [
13
+ "crewai-memory-core>=0.1.0",
14
+ "azure-cosmos>=4.7",
15
+ ]
16
+ keywords = ["crewai", "memory", "storage-backend", "cosmosdb", "azure", "long-term-memory", "agent-memory", "vector-search"]
17
+ classifiers = [
18
+ "Programming Language :: Python :: 3",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Operating System :: OS Independent",
21
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
22
+ ]
23
+
24
+ [project.urls]
25
+ Homepage = "https://github.com/skamalj/crewai-memory"
26
+ Repository = "https://github.com/skamalj/crewai-memory.git"
27
+ Documentation = "https://skamalj.github.io/agentstate-reducer/"
28
+
29
+ [dependency-groups]
30
+ dev = ["pytest>=7.0", "pytest-asyncio"]
31
+
32
+ [tool.pytest.ini_options]
33
+ asyncio_mode = "auto"
34
+
35
+ [tool.hatch.build.targets.wheel]
36
+ packages = ["src/crewai_memory_cosmosdb"]
@@ -0,0 +1,141 @@
1
+ """Azure Cosmos DB (NoSQL) ``StorageBackend`` for CrewAI unified ``Memory``.
2
+
3
+ A container partitioned by ``/scope`` holding one document per ``MemoryRecord``
4
+ (``id`` = record id) with a **vector embedding policy** on ``/embedding``
5
+ (cosine) and a ``diskANN`` vector index. ``search`` is a single SQL query:
6
+ ``ORDER BY VectorDistance(c.embedding, @q)`` with the scope subtree filtered in
7
+ ``WHERE``. Built on ``crewai-memory-core``.
8
+
9
+ The account needs the ``EnableNoSQLVectorSearch`` capability; the vector policy
10
+ can only be set at container creation.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from typing import Any, List, Optional, Tuple
16
+
17
+ from azure.cosmos import CosmosClient, PartitionKey
18
+ from azure.cosmos.exceptions import CosmosResourceNotFoundError
19
+
20
+ from crewai_memory_core import MemoryBackend, ROW_FIELDS, norm_scope
21
+
22
+ __all__ = ["CosmosDBMemoryBackend"]
23
+
24
+ _PROJ = ", ".join(f'c["{f}"]' for f in ROW_FIELDS if f != "embedding")
25
+
26
+
27
+ class CosmosDBMemoryBackend(MemoryBackend):
28
+ """CrewAI ``StorageBackend`` on Azure Cosmos DB with native vector search.
29
+
30
+ Example:
31
+ ```python
32
+ from crewai.memory import Memory
33
+ from crewai_memory_cosmosdb import CosmosDBMemoryBackend
34
+
35
+ backend = CosmosDBMemoryBackend(endpoint=..., key=..., database_name="crewai",
36
+ container_name="memory", dimensions=3072)
37
+ memory = Memory(storage=backend)
38
+ ```
39
+ """
40
+
41
+ supports_native_vector_search = True
42
+
43
+ def __init__(
44
+ self,
45
+ *,
46
+ endpoint: str,
47
+ key: str,
48
+ database_name: str,
49
+ container_name: str,
50
+ dimensions: int = 3072,
51
+ vector_index_type: str = "diskANN",
52
+ ) -> None:
53
+ self.dimensions = dimensions
54
+ client = CosmosClient(endpoint, credential=key)
55
+ db = client.create_database_if_not_exists(database_name)
56
+ self._container = db.create_container_if_not_exists(
57
+ id=container_name,
58
+ partition_key=PartitionKey(path="/scope"),
59
+ vector_embedding_policy={
60
+ "vectorEmbeddings": [
61
+ {"path": "/embedding", "dataType": "float32", "distanceFunction": "cosine", "dimensions": dimensions}
62
+ ]
63
+ },
64
+ indexing_policy={
65
+ "automatic": True,
66
+ "indexingMode": "consistent",
67
+ "includedPaths": [{"path": "/*"}],
68
+ "excludedPaths": [{"path": "/embedding/*"}, {"path": '/"_etag"/?'}],
69
+ "vectorIndexes": [{"path": "/embedding", "type": vector_index_type}],
70
+ },
71
+ )
72
+
73
+ # ------------------------------------------------------------------ helpers
74
+ @staticmethod
75
+ def _row(doc: dict) -> dict:
76
+ row = {f: doc.get(f) for f in ROW_FIELDS}
77
+ row["scope"] = norm_scope(doc.get("scope"))
78
+ return row
79
+
80
+ @staticmethod
81
+ def _scope_where(scope_prefix: Optional[str]) -> Tuple[str, list]:
82
+ p = norm_scope(scope_prefix)
83
+ if p == "/":
84
+ return "true", []
85
+ return '(c["scope"] = @p OR STARTSWITH(c["scope"], @pc))', [
86
+ {"name": "@p", "value": p}, {"name": "@pc", "value": p + "/"},
87
+ ]
88
+
89
+ def _query(self, sql: str, params: list) -> List[dict]:
90
+ return list(self._container.query_items(query=sql, parameters=params, enable_cross_partition_query=True))
91
+
92
+ # ------------------------------------------------------------------ primitives
93
+ def _put(self, rows: List[dict]) -> None:
94
+ for row in rows:
95
+ existing = self._get(row["id"])
96
+ if existing and existing["scope"] != norm_scope(row["scope"]):
97
+ self._delete_doc(row["id"], existing["scope"]) # scope moved: partition changes
98
+ doc = {f: row.get(f) for f in ROW_FIELDS}
99
+ doc["scope"] = norm_scope(row["scope"])
100
+ if not doc.get("embedding"):
101
+ doc.pop("embedding", None)
102
+ self._container.upsert_item(doc)
103
+
104
+ def _get(self, record_id: str) -> Optional[dict]:
105
+ docs = self._query('SELECT TOP 1 * FROM c WHERE c["id"] = @id', [{"name": "@id", "value": record_id}])
106
+ return self._row(docs[0]) if docs else None
107
+
108
+ def _delete_doc(self, record_id: str, scope: str) -> bool:
109
+ try:
110
+ self._container.delete_item(item=record_id, partition_key=scope)
111
+ return True
112
+ except CosmosResourceNotFoundError:
113
+ return False
114
+
115
+ def _delete_ids(self, ids: List[str]) -> int:
116
+ n = 0
117
+ for rid in ids:
118
+ row = self._get(rid)
119
+ if row and self._delete_doc(rid, row["scope"]):
120
+ n += 1
121
+ return n
122
+
123
+ def _scan(self, scope_prefix: Optional[str]) -> List[dict]:
124
+ where, params = self._scope_where(scope_prefix)
125
+ return [self._row(d) for d in self._query(f"SELECT * FROM c WHERE {where}", params)]
126
+
127
+ def _vector_search(
128
+ self, vector: List[float], scope_prefix: Optional[str], limit: int
129
+ ) -> List[Tuple[dict, float]]:
130
+ where, params = self._scope_where(scope_prefix)
131
+ sql = (
132
+ f'SELECT TOP @k {_PROJ}, VectorDistance(c["embedding"], @q) AS score '
133
+ f'FROM c WHERE {where} AND IS_DEFINED(c["embedding"]) '
134
+ f'ORDER BY VectorDistance(c["embedding"], @q)'
135
+ )
136
+ params = params + [{"name": "@q", "value": vector}, {"name": "@k", "value": int(limit)}]
137
+ out: List[Tuple[dict, float]] = []
138
+ for d in self._query(sql, params):
139
+ row = self._row(d)
140
+ out.append((row, float(d["score"]))) # cosine similarity, higher is better
141
+ return out
@@ -0,0 +1,46 @@
1
+ """Cosmos DB StorageBackend: the shared contract suite against a real Cosmos account."""
2
+ import os
3
+ import uuid
4
+
5
+ import pytest
6
+
7
+ from crewai_memory_core.contract import * # noqa: F401,F403
8
+ from crewai_memory_core.contract import DIMS
9
+ from crewai_memory_cosmosdb import CosmosDBMemoryBackend
10
+
11
+ EP = os.environ.get("COSMOS_ENDPOINT")
12
+ KEY = os.environ.get("COSMOS_KEY")
13
+ DB = os.environ.get("CREWAI_COSMOS_DB", "crewai_memory_test")
14
+
15
+ pytestmark = pytest.mark.skipif(not (EP and KEY), reason="COSMOS_ENDPOINT/COSMOS_KEY not set")
16
+
17
+
18
+ @pytest.fixture(scope="module")
19
+ def _store():
20
+ cname = f"mem_{uuid.uuid4().hex[:8]}"
21
+ b = CosmosDBMemoryBackend(endpoint=EP, key=KEY, database_name=DB, container_name=cname, dimensions=DIMS)
22
+ yield b
23
+ from azure.cosmos import CosmosClient
24
+ CosmosClient(EP, credential=KEY).get_database_client(DB).delete_container(cname)
25
+
26
+
27
+ @pytest.fixture()
28
+ def backend(_store):
29
+ _store.reset()
30
+ yield _store
31
+ _store.reset()
32
+
33
+
34
+ def test_native_vector_path_used(backend, monkeypatch):
35
+ from crewai_memory_core.contract import _seed, EMB
36
+ _seed(backend)
37
+ seen = []
38
+ orig = backend._container.query_items
39
+
40
+ def spy(query, **kw):
41
+ seen.append(query)
42
+ return orig(query=query, **kw)
43
+
44
+ monkeypatch.setattr(backend._container, "query_items", spy)
45
+ assert backend.search(EMB(["sushi"])[0], scope_prefix="/crew", limit=5)
46
+ assert any("VectorDistance" in q for q in seen)