patternmem-rag 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- patternmem/__init__.py +40 -0
- patternmem/_utils.py +25 -0
- patternmem/augmenter.py +155 -0
- patternmem/backend.py +138 -0
- patternmem/backends/__init__.py +1 -0
- patternmem/backends/chroma_backend.py +215 -0
- patternmem/backends/faiss_backend.py +302 -0
- patternmem/backends/json_backend.py +173 -0
- patternmem/backends/neo4j_backend.py +236 -0
- patternmem/backends/networkx_backend.py +157 -0
- patternmem/backends/sqlite_backend.py +197 -0
- patternmem/decay.py +90 -0
- patternmem/eval_router.py +316 -0
- patternmem/middleware.py +336 -0
- patternmem/observability.py +101 -0
- patternmem/reflector.py +211 -0
- patternmem/resolver.py +138 -0
- patternmem/types.py +212 -0
- patternmem_rag-0.1.1.dist-info/METADATA +238 -0
- patternmem_rag-0.1.1.dist-info/RECORD +23 -0
- patternmem_rag-0.1.1.dist-info/WHEEL +5 -0
- patternmem_rag-0.1.1.dist-info/licenses/LICENSE +21 -0
- patternmem_rag-0.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
"""
|
|
2
|
+
patternmem.backends.faiss_backend
|
|
3
|
+
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
4
|
+
FAISS (Facebook AI Similarity Search) backend for PatternMem.
|
|
5
|
+
|
|
6
|
+
Uses a ``faiss.IndexFlatIP`` (inner-product) index on L2-normalised embeddings,
|
|
7
|
+
which is mathematically equivalent to cosine similarity search. FAISS is
|
|
8
|
+
extremely fast for large pattern memories and runs fully locally with no server.
|
|
9
|
+
|
|
10
|
+
Storage layout
|
|
11
|
+
--------------
|
|
12
|
+
Two files, both in the same directory:
|
|
13
|
+
|
|
14
|
+
- ``<name>.index`` — the FAISS binary index (embeddings only, sequential rows)
|
|
15
|
+
- ``<name>.meta.json`` — JSON sidecar: a list of FailurePattern dicts in the
|
|
16
|
+
same row order as the FAISS index.
|
|
17
|
+
|
|
18
|
+
Row positions in the FAISS index map 1-to-1 with list indices in the sidecar.
|
|
19
|
+
On delete/update, the sidecar entry is marked ``null`` and the row is
|
|
20
|
+
logically ignored during lookup (a compaction rebuilds the index if needed).
|
|
21
|
+
|
|
22
|
+
Why not IndexIDMap?
|
|
23
|
+
-------------------
|
|
24
|
+
``faiss.IndexIDMap`` has a known segfault on macOS Apple Silicon with certain
|
|
25
|
+
builds. Using a plain ``IndexFlatIP`` plus a positional sidecar avoids
|
|
26
|
+
the issue entirely while keeping the same search semantics.
|
|
27
|
+
|
|
28
|
+
Similarity
|
|
29
|
+
----------
|
|
30
|
+
``IndexFlatIP`` returns the exact inner product. Since embeddings are
|
|
31
|
+
L2-normalised by MiniLM before being passed to PatternMem, inner product ==
|
|
32
|
+
cosine similarity. The ``similarity_threshold`` is applied as a post-filter.
|
|
33
|
+
|
|
34
|
+
Requirements
|
|
35
|
+
------------
|
|
36
|
+
pip install patternmem-rag[faiss]
|
|
37
|
+
# or
|
|
38
|
+
pip install faiss-cpu>=1.7 # CPU-only (recommended)
|
|
39
|
+
# pip install faiss-gpu>=1.7 # GPU version (if CUDA is available)
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
import asyncio
|
|
45
|
+
import json
|
|
46
|
+
from datetime import datetime, timezone
|
|
47
|
+
from pathlib import Path
|
|
48
|
+
from typing import Any
|
|
49
|
+
|
|
50
|
+
import numpy as np
|
|
51
|
+
|
|
52
|
+
from patternmem.backend import MemoryBackend
|
|
53
|
+
from patternmem.types import FailurePattern, FailureType
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
import faiss # type: ignore[import]
|
|
57
|
+
|
|
58
|
+
_FAISS_AVAILABLE = True
|
|
59
|
+
except ImportError:
|
|
60
|
+
_FAISS_AVAILABLE = False
|
|
61
|
+
|
|
62
|
+
# Default directory for FAISS index files
|
|
63
|
+
_DEFAULT_DIR = Path.home() / ".patternmem"
|
|
64
|
+
_DEFAULT_NAME = "patterns_faiss"
|
|
65
|
+
|
|
66
|
+
# MiniLM embedding dimension (must match PatternMemMiddleware._EMBED_MODEL)
|
|
67
|
+
_DIM = 384
|
|
68
|
+
|
|
69
|
+
# Sentinel for a deleted/unused row in the sidecar
|
|
70
|
+
_DELETED: None = None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _to_float32(embedding: list[float]) -> "np.ndarray[Any, np.dtype[np.float32]]":
|
|
74
|
+
"""Convert a list of floats to a normalised float32 row-vector."""
|
|
75
|
+
arr = np.array(embedding, dtype=np.float32)
|
|
76
|
+
norm = float(np.linalg.norm(arr))
|
|
77
|
+
if norm > 0:
|
|
78
|
+
arr /= norm
|
|
79
|
+
return arr.reshape(1, -1)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _pattern_to_dict(p: FailurePattern) -> dict[str, Any]:
|
|
83
|
+
return {
|
|
84
|
+
"id": p.id,
|
|
85
|
+
"query_embedding": p.query_embedding,
|
|
86
|
+
"failure_type": p.failure_type.name,
|
|
87
|
+
"root_cause": p.root_cause,
|
|
88
|
+
"hint_text": p.hint_text,
|
|
89
|
+
"score": p.score,
|
|
90
|
+
"created_at": p.created_at.isoformat(),
|
|
91
|
+
"decay_weight": p.decay_weight,
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _dict_to_pattern(d: dict[str, Any]) -> FailurePattern:
|
|
96
|
+
return FailurePattern(
|
|
97
|
+
id=d["id"],
|
|
98
|
+
query_embedding=d["query_embedding"],
|
|
99
|
+
failure_type=FailureType[d["failure_type"]],
|
|
100
|
+
root_cause=d["root_cause"],
|
|
101
|
+
hint_text=d["hint_text"],
|
|
102
|
+
score=float(d["score"]),
|
|
103
|
+
created_at=datetime.fromisoformat(d["created_at"]).replace(
|
|
104
|
+
tzinfo=timezone.utc
|
|
105
|
+
),
|
|
106
|
+
decay_weight=float(d["decay_weight"]),
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class FAISSBackend(MemoryBackend):
|
|
111
|
+
"""FAISS-backed vector index for PatternMem failure patterns.
|
|
112
|
+
|
|
113
|
+
Extremely fast exact nearest-neighbour search, fully local, no server.
|
|
114
|
+
Best for large pattern memories where brute-force numpy becomes slow.
|
|
115
|
+
|
|
116
|
+
Uses a plain ``faiss.IndexFlatIP`` (no IndexIDMap) for maximum
|
|
117
|
+
compatibility across platforms, including macOS Apple Silicon.
|
|
118
|
+
|
|
119
|
+
Parameters
|
|
120
|
+
----------
|
|
121
|
+
directory:
|
|
122
|
+
Directory where the index (``.index``) and metadata (``.meta.json``)
|
|
123
|
+
files are stored. Created automatically if absent.
|
|
124
|
+
name:
|
|
125
|
+
Base name for the two files. Defaults to ``"patterns_faiss"``.
|
|
126
|
+
similarity_threshold:
|
|
127
|
+
Minimum cosine similarity for a pattern to be returned.
|
|
128
|
+
compaction_threshold:
|
|
129
|
+
When the fraction of deleted (null) rows in the index exceeds this
|
|
130
|
+
value, a full compaction is triggered on the next write.
|
|
131
|
+
Default: 0.3 (compact when 30 %+ of rows are tombstones).
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
def __init__(
|
|
135
|
+
self,
|
|
136
|
+
directory: str | Path = _DEFAULT_DIR,
|
|
137
|
+
name: str = _DEFAULT_NAME,
|
|
138
|
+
similarity_threshold: float = 0.82,
|
|
139
|
+
compaction_threshold: float = 0.3,
|
|
140
|
+
) -> None:
|
|
141
|
+
if not _FAISS_AVAILABLE:
|
|
142
|
+
raise ImportError(
|
|
143
|
+
"FAISSBackend requires 'faiss-cpu' (or 'faiss-gpu'). "
|
|
144
|
+
"Install it with: pip install patternmem-rag[faiss]"
|
|
145
|
+
)
|
|
146
|
+
self._threshold = similarity_threshold
|
|
147
|
+
self._compaction_threshold = compaction_threshold
|
|
148
|
+
self._dir = Path(directory)
|
|
149
|
+
self._index_path = self._dir / f"{name}.index"
|
|
150
|
+
self._meta_path = self._dir / f"{name}.meta.json"
|
|
151
|
+
self._lock = asyncio.Lock()
|
|
152
|
+
|
|
153
|
+
# sidecar: list of (pattern_dict | None)
|
|
154
|
+
# None = row deleted / not in use
|
|
155
|
+
self._sidecar: list[dict[str, Any] | None] = self._load_meta()
|
|
156
|
+
|
|
157
|
+
# Rebuild index from sidecar on startup
|
|
158
|
+
self._index: Any = self._build_index_from_sidecar()
|
|
159
|
+
|
|
160
|
+
# id_to_row: pattern UUID → row index in the sidecar/index
|
|
161
|
+
self._id_to_row: dict[str, int] = {
|
|
162
|
+
d["id"]: i
|
|
163
|
+
for i, d in enumerate(self._sidecar)
|
|
164
|
+
if d is not None
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
# ------------------------------------------------------------------
|
|
168
|
+
# Persistence helpers
|
|
169
|
+
# ------------------------------------------------------------------
|
|
170
|
+
|
|
171
|
+
def _ensure_dir(self) -> None:
|
|
172
|
+
self._dir.mkdir(parents=True, exist_ok=True)
|
|
173
|
+
|
|
174
|
+
def _load_meta(self) -> list[dict[str, Any] | None]:
|
|
175
|
+
if not self._meta_path.exists():
|
|
176
|
+
return []
|
|
177
|
+
with open(self._meta_path, "r", encoding="utf-8") as fh:
|
|
178
|
+
try:
|
|
179
|
+
raw = json.load(fh)
|
|
180
|
+
return raw # list of dicts or nulls
|
|
181
|
+
except json.JSONDecodeError:
|
|
182
|
+
return []
|
|
183
|
+
|
|
184
|
+
def _build_index_from_sidecar(self) -> Any:
|
|
185
|
+
"""Rebuild a fresh IndexFlatIP from all non-null sidecar rows."""
|
|
186
|
+
idx = faiss.IndexFlatIP(_DIM)
|
|
187
|
+
active = [d for d in self._sidecar if d is not None]
|
|
188
|
+
if active:
|
|
189
|
+
vecs = np.vstack(
|
|
190
|
+
[_to_float32(d["query_embedding"]) for d in active]
|
|
191
|
+
).astype(np.float32)
|
|
192
|
+
idx.add(vecs)
|
|
193
|
+
return idx
|
|
194
|
+
|
|
195
|
+
def _save(self) -> None:
|
|
196
|
+
"""Atomically persist the FAISS index and metadata sidecar."""
|
|
197
|
+
self._ensure_dir()
|
|
198
|
+
faiss.write_index(self._index, str(self._index_path))
|
|
199
|
+
tmp = self._meta_path.with_suffix(".tmp")
|
|
200
|
+
with open(tmp, "w", encoding="utf-8") as fh:
|
|
201
|
+
json.dump(self._sidecar, fh, indent=2)
|
|
202
|
+
tmp.replace(self._meta_path)
|
|
203
|
+
|
|
204
|
+
def _compact(self) -> None:
|
|
205
|
+
"""Remove all null rows, rebuild the index and reset id_to_row."""
|
|
206
|
+
active = [(i, d) for i, d in enumerate(self._sidecar) if d is not None]
|
|
207
|
+
self._sidecar = [d for _, d in active]
|
|
208
|
+
self._index = self._build_index_from_sidecar()
|
|
209
|
+
self._id_to_row = {d["id"]: new_i for new_i, d in enumerate(self._sidecar)}
|
|
210
|
+
|
|
211
|
+
def _should_compact(self) -> bool:
|
|
212
|
+
total = len(self._sidecar)
|
|
213
|
+
if total == 0:
|
|
214
|
+
return False
|
|
215
|
+
nulls = sum(1 for d in self._sidecar if d is None)
|
|
216
|
+
return nulls / total >= self._compaction_threshold
|
|
217
|
+
|
|
218
|
+
# ------------------------------------------------------------------
|
|
219
|
+
# MemoryBackend implementation
|
|
220
|
+
# ------------------------------------------------------------------
|
|
221
|
+
|
|
222
|
+
async def write_pattern(self, pattern: FailurePattern) -> None:
|
|
223
|
+
async with self._lock:
|
|
224
|
+
pid = pattern.id
|
|
225
|
+
|
|
226
|
+
if pid in self._id_to_row:
|
|
227
|
+
# Mark old row as deleted (tombstone), then append fresh entry
|
|
228
|
+
old_row = self._id_to_row[pid]
|
|
229
|
+
self._sidecar[old_row] = None
|
|
230
|
+
del self._id_to_row[pid]
|
|
231
|
+
|
|
232
|
+
if self._should_compact():
|
|
233
|
+
self._compact()
|
|
234
|
+
|
|
235
|
+
# Append new vector and metadata
|
|
236
|
+
vec = _to_float32(pattern.query_embedding)
|
|
237
|
+
self._index.add(vec)
|
|
238
|
+
new_row = len(self._sidecar)
|
|
239
|
+
self._sidecar.append(_pattern_to_dict(pattern))
|
|
240
|
+
self._id_to_row[pid] = new_row
|
|
241
|
+
self._save()
|
|
242
|
+
|
|
243
|
+
async def lookup_patterns(
|
|
244
|
+
self,
|
|
245
|
+
query_embedding: list[float],
|
|
246
|
+
top_k: int = 3,
|
|
247
|
+
) -> list[FailurePattern]:
|
|
248
|
+
async with self._lock:
|
|
249
|
+
active_count = self._index.ntotal
|
|
250
|
+
if active_count == 0:
|
|
251
|
+
return []
|
|
252
|
+
|
|
253
|
+
vec = _to_float32(query_embedding)
|
|
254
|
+
k = min(top_k * 4, active_count)
|
|
255
|
+
scores_arr, rows_arr = self._index.search(vec, k)
|
|
256
|
+
|
|
257
|
+
# Map FAISS sequential row positions back to sidecar positions
|
|
258
|
+
# After compaction, FAISS rows align with non-null sidecar entries
|
|
259
|
+
active_rows = [i for i, d in enumerate(self._sidecar) if d is not None]
|
|
260
|
+
|
|
261
|
+
results: list[tuple[float, FailurePattern]] = []
|
|
262
|
+
for faiss_pos, score in zip(rows_arr[0].tolist(), scores_arr[0].tolist()):
|
|
263
|
+
if faiss_pos < 0 or faiss_pos >= len(active_rows):
|
|
264
|
+
continue
|
|
265
|
+
sidecar_idx = active_rows[faiss_pos]
|
|
266
|
+
d = self._sidecar[sidecar_idx]
|
|
267
|
+
if d is None:
|
|
268
|
+
continue
|
|
269
|
+
similarity = float(score)
|
|
270
|
+
if similarity >= self._threshold:
|
|
271
|
+
results.append((similarity, _dict_to_pattern(d)))
|
|
272
|
+
|
|
273
|
+
results.sort(key=lambda t: t[0], reverse=True)
|
|
274
|
+
return [p for _, p in results[:top_k]]
|
|
275
|
+
|
|
276
|
+
async def get_stats(self) -> dict[str, Any]:
|
|
277
|
+
async with self._lock:
|
|
278
|
+
active = sum(1 for d in self._sidecar if d is not None)
|
|
279
|
+
return {
|
|
280
|
+
"count": active,
|
|
281
|
+
"index_path": str(self._index_path),
|
|
282
|
+
"backend": "faiss",
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
async def update_pattern(self, pattern_id: str, decay_weight: float) -> None:
|
|
286
|
+
async with self._lock:
|
|
287
|
+
if pattern_id not in self._id_to_row:
|
|
288
|
+
raise KeyError(f"Pattern {pattern_id!r} not found in FAISS backend")
|
|
289
|
+
row = self._id_to_row[pattern_id]
|
|
290
|
+
if self._sidecar[row] is not None:
|
|
291
|
+
self._sidecar[row]["decay_weight"] = decay_weight # type: ignore[index]
|
|
292
|
+
self._save()
|
|
293
|
+
|
|
294
|
+
async def delete_pattern(self, pattern_id: str) -> None:
|
|
295
|
+
async with self._lock:
|
|
296
|
+
if pattern_id not in self._id_to_row:
|
|
297
|
+
raise KeyError(f"Pattern {pattern_id!r} not found in FAISS backend")
|
|
298
|
+
row = self._id_to_row.pop(pattern_id)
|
|
299
|
+
self._sidecar[row] = None
|
|
300
|
+
# Compact immediately on delete to keep the index honest
|
|
301
|
+
self._compact()
|
|
302
|
+
self._save()
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""
|
|
2
|
+
patternmem.backends.json_backend
|
|
3
|
+
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
4
|
+
Zero-dependency JSON file backend for PatternMem.
|
|
5
|
+
|
|
6
|
+
This is the **first-class** backend for zero-credential mode
|
|
7
|
+
(``backend="json"``, ``eval="none"``, ``observability=None``).
|
|
8
|
+
It requires only stdlib + numpy (already a sentence-transformers transitive dep).
|
|
9
|
+
|
|
10
|
+
Storage layout
|
|
11
|
+
--------------
|
|
12
|
+
A single JSON file, default ``~/.patternmem/patterns.json``. The file is an
|
|
13
|
+
object mapping ``pattern_id → serialised FailurePattern``.
|
|
14
|
+
|
|
15
|
+
Concurrency
|
|
16
|
+
-----------
|
|
17
|
+
An ``asyncio.Lock`` serialises all file reads and writes. This is sufficient
|
|
18
|
+
for a single-process use case; for multi-process workloads use SQLite or Neo4j.
|
|
19
|
+
|
|
20
|
+
Similarity
|
|
21
|
+
----------
|
|
22
|
+
Cosine similarity via numpy. Brute-force over all stored embeddings —
|
|
23
|
+
acceptable for the typical pattern-memory sizes (< 10 000 patterns).
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import asyncio
|
|
29
|
+
import json
|
|
30
|
+
import os
|
|
31
|
+
from datetime import datetime, timezone
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Any
|
|
34
|
+
|
|
35
|
+
from patternmem._utils import cosine_similarity as _cosine_similarity
|
|
36
|
+
from patternmem.backend import MemoryBackend
|
|
37
|
+
from patternmem.types import FailurePattern, FailureType
|
|
38
|
+
|
|
39
|
+
# Default storage path
|
|
40
|
+
_DEFAULT_PATH = Path.home() / ".patternmem" / "patterns.json"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _pattern_to_dict(p: FailurePattern) -> dict[str, Any]:
|
|
45
|
+
return {
|
|
46
|
+
"id": p.id,
|
|
47
|
+
"query_embedding": p.query_embedding,
|
|
48
|
+
"failure_type": p.failure_type.name,
|
|
49
|
+
"root_cause": p.root_cause,
|
|
50
|
+
"hint_text": p.hint_text,
|
|
51
|
+
"score": p.score,
|
|
52
|
+
"created_at": p.created_at.isoformat(),
|
|
53
|
+
"decay_weight": p.decay_weight,
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _dict_to_pattern(d: dict[str, Any]) -> FailurePattern:
|
|
58
|
+
return FailurePattern(
|
|
59
|
+
id=d["id"],
|
|
60
|
+
query_embedding=d["query_embedding"],
|
|
61
|
+
failure_type=FailureType[d["failure_type"]],
|
|
62
|
+
root_cause=d["root_cause"],
|
|
63
|
+
hint_text=d["hint_text"],
|
|
64
|
+
score=d["score"],
|
|
65
|
+
created_at=datetime.fromisoformat(d["created_at"]).replace(tzinfo=timezone.utc),
|
|
66
|
+
decay_weight=d["decay_weight"],
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class JSONBackend(MemoryBackend):
|
|
71
|
+
"""File-backed JSON storage — zero external dependencies beyond numpy.
|
|
72
|
+
|
|
73
|
+
Parameters
|
|
74
|
+
----------
|
|
75
|
+
path:
|
|
76
|
+
Path to the JSON file. Created automatically if it does not exist.
|
|
77
|
+
similarity_threshold:
|
|
78
|
+
Minimum cosine similarity for a pattern to be returned by
|
|
79
|
+
``lookup_patterns``. Should match ``PatternMemMiddleware``'s value.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
def __init__(
|
|
83
|
+
self,
|
|
84
|
+
path: str | Path = _DEFAULT_PATH,
|
|
85
|
+
similarity_threshold: float = 0.82,
|
|
86
|
+
) -> None:
|
|
87
|
+
self._path = Path(path)
|
|
88
|
+
self._threshold = similarity_threshold
|
|
89
|
+
self._lock = asyncio.Lock()
|
|
90
|
+
|
|
91
|
+
# ------------------------------------------------------------------
|
|
92
|
+
# Internal helpers
|
|
93
|
+
# ------------------------------------------------------------------
|
|
94
|
+
|
|
95
|
+
def _ensure_dir(self) -> None:
|
|
96
|
+
self._path.parent.mkdir(parents=True, exist_ok=True)
|
|
97
|
+
|
|
98
|
+
def _read_all(self) -> dict[str, dict[str, Any]]:
|
|
99
|
+
"""Load the JSON file; return empty dict if file does not exist."""
|
|
100
|
+
if not self._path.exists():
|
|
101
|
+
return {}
|
|
102
|
+
with open(self._path, "r", encoding="utf-8") as fh:
|
|
103
|
+
try:
|
|
104
|
+
data: dict[str, dict[str, Any]] = json.load(fh)
|
|
105
|
+
except json.JSONDecodeError:
|
|
106
|
+
return {}
|
|
107
|
+
return data
|
|
108
|
+
|
|
109
|
+
def _write_all(self, data: dict[str, dict[str, Any]]) -> None:
|
|
110
|
+
self._ensure_dir()
|
|
111
|
+
tmp = self._path.with_suffix(".tmp")
|
|
112
|
+
with open(tmp, "w", encoding="utf-8") as fh:
|
|
113
|
+
json.dump(data, fh, indent=2, ensure_ascii=False)
|
|
114
|
+
# Atomic replace
|
|
115
|
+
tmp.replace(self._path)
|
|
116
|
+
|
|
117
|
+
# ------------------------------------------------------------------
|
|
118
|
+
# MemoryBackend implementation
|
|
119
|
+
# ------------------------------------------------------------------
|
|
120
|
+
|
|
121
|
+
async def write_pattern(self, pattern: FailurePattern) -> None:
|
|
122
|
+
async with self._lock:
|
|
123
|
+
data = self._read_all()
|
|
124
|
+
data[pattern.id] = _pattern_to_dict(pattern)
|
|
125
|
+
self._write_all(data)
|
|
126
|
+
|
|
127
|
+
async def lookup_patterns(
|
|
128
|
+
self,
|
|
129
|
+
query_embedding: list[float],
|
|
130
|
+
top_k: int = 3,
|
|
131
|
+
) -> list[FailurePattern]:
|
|
132
|
+
async with self._lock:
|
|
133
|
+
data = self._read_all()
|
|
134
|
+
|
|
135
|
+
if not data:
|
|
136
|
+
return []
|
|
137
|
+
|
|
138
|
+
scored: list[tuple[float, FailurePattern]] = []
|
|
139
|
+
for raw in data.values():
|
|
140
|
+
stored_emb: list[float] = raw["query_embedding"]
|
|
141
|
+
if not stored_emb:
|
|
142
|
+
continue
|
|
143
|
+
sim = _cosine_similarity(query_embedding, stored_emb)
|
|
144
|
+
if sim >= self._threshold:
|
|
145
|
+
scored.append((sim, _dict_to_pattern(raw)))
|
|
146
|
+
|
|
147
|
+
scored.sort(key=lambda t: t[0], reverse=True)
|
|
148
|
+
return [p for _, p in scored[:top_k]]
|
|
149
|
+
|
|
150
|
+
async def get_stats(self) -> dict[str, Any]:
|
|
151
|
+
async with self._lock:
|
|
152
|
+
data = self._read_all()
|
|
153
|
+
return {
|
|
154
|
+
"count": len(data),
|
|
155
|
+
"path": str(self._path),
|
|
156
|
+
"backend": "json",
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
async def update_pattern(self, pattern_id: str, decay_weight: float) -> None:
|
|
160
|
+
async with self._lock:
|
|
161
|
+
data = self._read_all()
|
|
162
|
+
if pattern_id not in data:
|
|
163
|
+
raise KeyError(f"Pattern {pattern_id!r} not found in JSON backend")
|
|
164
|
+
data[pattern_id]["decay_weight"] = decay_weight
|
|
165
|
+
self._write_all(data)
|
|
166
|
+
|
|
167
|
+
async def delete_pattern(self, pattern_id: str) -> None:
|
|
168
|
+
async with self._lock:
|
|
169
|
+
data = self._read_all()
|
|
170
|
+
if pattern_id not in data:
|
|
171
|
+
raise KeyError(f"Pattern {pattern_id!r} not found in JSON backend")
|
|
172
|
+
del data[pattern_id]
|
|
173
|
+
self._write_all(data)
|