engram-memory-system 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
engram/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Engram: Cognitive memory system with hybrid retrieval."""
2
+
3
+ __version__ = "0.1.0"
engram/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from engram.cli import main
2
+
3
+ main()
engram/ann_index.py ADDED
@@ -0,0 +1,210 @@
1
+ """HNSW approximate nearest neighbor index for fast dense retrieval.
2
+
3
+ Wraps hnswlib to replace brute-force cosine similarity search.
4
+ At 200k vectors, drops dense search from ~5s to <10ms.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import logging
10
+ import threading
11
+ import time
12
+ from pathlib import Path
13
+
14
+ import numpy as np
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ try:
19
+ import hnswlib
20
+ HAS_HNSWLIB = True
21
+ except ImportError:
22
+ HAS_HNSWLIB = False
23
+
24
+
25
+ class ANNIndex:
26
+ """HNSW index for approximate cosine nearest-neighbor search."""
27
+
28
+ def __init__(self, dim: int = 384, m: int = 32, ef_construction: int = 200,
29
+ ef_search: int = 100, max_elements: int = 500_000,
30
+ index_path: str | None = None):
31
+ if not HAS_HNSWLIB:
32
+ raise ImportError("hnswlib not installed: pip install hnswlib")
33
+
34
+ self.dim = dim
35
+ self.m = m
36
+ self.ef_construction = ef_construction
37
+ self.ef_search = ef_search
38
+ self.max_elements = max_elements
39
+ self.index_path = Path(index_path) if index_path else None
40
+
41
+ self._index: hnswlib.Index | None = None
42
+ self._id_to_label: dict[str, int] = {} # memory_id → hnswlib label
43
+ self._label_to_id: dict[int, str] = {} # hnswlib label → memory_id
44
+ self._next_label: int = 0
45
+ self._lock = threading.Lock()
46
+ self._ready = False
47
+
48
+ @property
49
+ def ready(self) -> bool:
50
+ return self._ready and self._index is not None
51
+
52
+ @property
53
+ def count(self) -> int:
54
+ """Active (non-deleted) vectors in the index."""
55
+ return len(self._id_to_label)
56
+
57
+ def build(self, ids: list[str], vecs: np.ndarray):
58
+ """Build index from scratch. ids and vecs must be aligned."""
59
+ if len(ids) == 0:
60
+ # init empty index so add() works later
61
+ idx = hnswlib.Index(space="cosine", dim=self.dim)
62
+ idx.init_index(max_elements=self.max_elements, M=self.m, ef_construction=self.ef_construction)
63
+ idx.set_ef(self.ef_search)
64
+ self._index = idx
65
+ self._ready = True
66
+ return
67
+
68
+ t0 = time.time()
69
+ n = len(ids)
70
+
71
+ idx = hnswlib.Index(space="cosine", dim=self.dim)
72
+ max_el = max(self.max_elements, n + 10_000)
73
+ idx.init_index(max_elements=max_el, M=self.m, ef_construction=self.ef_construction)
74
+ idx.set_ef(self.ef_search)
75
+
76
+ # add all vectors with integer labels
77
+ labels = np.arange(n, dtype=np.int64)
78
+ idx.add_items(vecs.astype(np.float32), labels, num_threads=4)
79
+
80
+ with self._lock:
81
+ self._index = idx
82
+ self._id_to_label = {mid: int(i) for i, mid in enumerate(ids)}
83
+ self._label_to_id = {int(i): mid for i, mid in enumerate(ids)}
84
+ self._next_label = n
85
+ self._ready = True
86
+
87
+ elapsed = time.time() - t0
88
+ logger.info(f"ANN index built: {n} vectors in {elapsed:.1f}s")
89
+
90
+ def add(self, memory_id: str, vec: np.ndarray):
91
+ """Add a single vector to the index. Thread-safe."""
92
+ if self._index is None:
93
+ return
94
+
95
+ with self._lock:
96
+ # if already exists, remove first
97
+ if memory_id in self._id_to_label:
98
+ old_label = self._id_to_label[memory_id]
99
+ self._index.mark_deleted(old_label)
100
+ del self._label_to_id[old_label]
101
+
102
+ label = self._next_label
103
+ self._next_label += 1
104
+
105
+ # resize if needed
106
+ if label >= self._index.get_max_elements():
107
+ new_max = self._index.get_max_elements() + 100_000
108
+ self._index.resize_index(new_max)
109
+
110
+ self._index.add_items(
111
+ vec.astype(np.float32).reshape(1, -1),
112
+ np.array([label], dtype=np.int64),
113
+ )
114
+ self._id_to_label[memory_id] = label
115
+ self._label_to_id[label] = memory_id
116
+
117
+ def remove(self, memory_id: str):
118
+ """Mark a vector as deleted. Thread-safe."""
119
+ if self._index is None:
120
+ return
121
+
122
+ with self._lock:
123
+ if memory_id in self._id_to_label:
124
+ label = self._id_to_label[memory_id]
125
+ self._index.mark_deleted(label)
126
+ del self._id_to_label[memory_id]
127
+ del self._label_to_id[label]
128
+
129
+ def search(self, query_vec: np.ndarray, top_k: int = 10) -> list[tuple[str, float]]:
130
+ """Search for nearest neighbors. Returns [(memory_id, similarity_score), ...].
131
+
132
+ Scores are cosine similarity (higher = more similar), matching
133
+ the convention of the brute-force fallback.
134
+ """
135
+ if not self._ready or self._index is None or self.count == 0:
136
+ return []
137
+
138
+ k = min(top_k, self.count)
139
+ q = query_vec.astype(np.float32).reshape(1, -1)
140
+
141
+ with self._lock:
142
+ labels, distances = self._index.knn_query(q, k=k)
143
+
144
+ results = []
145
+ for label, dist in zip(labels[0], distances[0]):
146
+ label = int(label)
147
+ if label in self._label_to_id:
148
+ # hnswlib cosine distance = 1 - cosine_similarity
149
+ similarity = 1.0 - float(dist)
150
+ results.append((self._label_to_id[label], similarity))
151
+
152
+ return results
153
+
154
+ def save(self, path: Path | str | None = None):
155
+ """Persist index to disk."""
156
+ if self._index is None:
157
+ return
158
+
159
+ save_path = Path(path) if path else self.index_path
160
+ if save_path is None:
161
+ return
162
+
163
+ save_path.parent.mkdir(parents=True, exist_ok=True)
164
+
165
+ with self._lock:
166
+ self._index.save_index(str(save_path))
167
+
168
+ # save id mappings alongside
169
+ import json
170
+ meta_path = save_path.with_suffix(".meta.json")
171
+ meta = {
172
+ "id_to_label": self._id_to_label,
173
+ "next_label": self._next_label,
174
+ "count": self._index.get_current_count(),
175
+ "saved_at": time.time(),
176
+ }
177
+ meta_path.write_text(json.dumps(meta))
178
+
179
+ logger.info(f"ANN index saved: {self.count} vectors → {save_path}")
180
+
181
+ def load(self, path: Path | str | None = None) -> bool:
182
+ """Load index from disk. Returns True if successful."""
183
+ load_path = Path(path) if path else self.index_path
184
+ if load_path is None or not load_path.exists():
185
+ return False
186
+
187
+ meta_path = load_path.with_suffix(".meta.json")
188
+ if not meta_path.exists():
189
+ return False
190
+
191
+ try:
192
+ import json
193
+ meta = json.loads(meta_path.read_text())
194
+
195
+ idx = hnswlib.Index(space="cosine", dim=self.dim)
196
+ idx.load_index(str(load_path), max_elements=self.max_elements)
197
+ idx.set_ef(self.ef_search)
198
+
199
+ with self._lock:
200
+ self._index = idx
201
+ self._id_to_label = meta["id_to_label"]
202
+ self._label_to_id = {int(v): k for k, v in self._id_to_label.items()}
203
+ self._next_label = meta["next_label"]
204
+ self._ready = True
205
+
206
+ logger.info(f"ANN index loaded: {self.count} vectors from {load_path}")
207
+ return True
208
+ except Exception as e:
209
+ logger.warning(f"Failed to load ANN index: {e}")
210
+ return False