contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,524 @@
|
|
|
1
|
+
"""Deterministic property-graph projection and bounded traversal."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import re
|
|
7
|
+
from collections import defaultdict, deque
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from uuid import UUID, uuid5
|
|
11
|
+
|
|
12
|
+
from contextos.core.enums import (
|
|
13
|
+
GraphNodeType,
|
|
14
|
+
GraphRelationType,
|
|
15
|
+
MemoryStatus,
|
|
16
|
+
)
|
|
17
|
+
from contextos.core.models import (
|
|
18
|
+
GraphCandidateEvidence,
|
|
19
|
+
GraphEdge,
|
|
20
|
+
GraphEdgeSupport,
|
|
21
|
+
GraphExpansion,
|
|
22
|
+
GraphNode,
|
|
23
|
+
GraphPath,
|
|
24
|
+
GraphPathEdge,
|
|
25
|
+
GraphPathNode,
|
|
26
|
+
Memory,
|
|
27
|
+
MemoryFilters,
|
|
28
|
+
)
|
|
29
|
+
from contextos.core.protocols import MemoryRepository, RelationRepository
|
|
30
|
+
from contextos.storage.graph_repo import SqliteGraphRepository
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
GRAPH_NAMESPACE = UUID("1e842c72-7e40-4f2a-b905-2f7e9bc10325")
|
|
34
|
+
_TOKEN = r"[A-Za-z][A-Za-z0-9+#._-]*"
|
|
35
|
+
_ENTITY = rf"(?:Project\s+)?{_TOKEN}(?:\s+{_TOKEN}){{0,2}}"
|
|
36
|
+
_NEGATED = re.compile(r"\b(?:does\s+not|do\s+not|did\s+not|never|no\s+longer|not)\b", re.I)
|
|
37
|
+
_RELATION = re.compile(
|
|
38
|
+
rf"\b(?P<source>{_ENTITY})\s+"
|
|
39
|
+
r"(?:(?:currently|now|previously|formerly)\s+)?"
|
|
40
|
+
r"(?P<verb>uses?|used|runs?|depends\s+on|is\s+part\s+of|belongs\s+to|works\s+on)\s+"
|
|
41
|
+
r"(?P<target>[^,;!?]+?)(?=[,;!?]|$)",
|
|
42
|
+
re.I,
|
|
43
|
+
)
|
|
44
|
+
_CURRENT_REPLACEMENT = re.compile(
|
|
45
|
+
rf"\b(?P<source>(?:Project\s+)?{_TOKEN})\s+used\s+.+?\s+previously\s+"
|
|
46
|
+
rf"but\s+now\s+uses\s+(?P<target>{_TOKEN}(?:\s+[A-Za-z0-9][A-Za-z0-9+#._-]*)?)",
|
|
47
|
+
re.I,
|
|
48
|
+
)
|
|
49
|
+
_KNOWN_TOOLS = {
|
|
50
|
+
"ollama", "llama.cpp", "vllm", "docker", "podman", "python", "rust",
|
|
51
|
+
"c++17", "qwen9b", "qwen30b", "qwen", "postgresql", "sqlite",
|
|
52
|
+
}
|
|
53
|
+
_STOP_LABELS = {
|
|
54
|
+
"user", "the", "a", "an", "current", "previous", "larger", "model",
|
|
55
|
+
"runtime", "tool", "memory", "project", "failed", "failure", "concise",
|
|
56
|
+
"which", "what", "when", "where", "why", "how",
|
|
57
|
+
}
|
|
58
|
+
_RELATION_MAP = {
|
|
59
|
+
"use": GraphRelationType.USES,
|
|
60
|
+
"uses": GraphRelationType.USES,
|
|
61
|
+
"used": GraphRelationType.USES,
|
|
62
|
+
"run": GraphRelationType.RUNS,
|
|
63
|
+
"runs": GraphRelationType.RUNS,
|
|
64
|
+
"depends on": GraphRelationType.DEPENDS_ON,
|
|
65
|
+
"is part of": GraphRelationType.PART_OF,
|
|
66
|
+
"belongs to": GraphRelationType.BELONGS_TO,
|
|
67
|
+
"works on": GraphRelationType.WORKS_ON,
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def canonical_key(label: str) -> str:
|
|
72
|
+
value = re.sub(r"^project\s+", "", label.strip(), flags=re.I)
|
|
73
|
+
value = " ".join(value.casefold().split())
|
|
74
|
+
return re.sub(r"\b(qwen)\s+(\d+b)\b", r"\1\2", value)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def stable_node_id(node_type: GraphNodeType, key: str) -> UUID:
|
|
78
|
+
return uuid5(GRAPH_NAMESPACE, f"node:{node_type.value}:{key}")
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def stable_edge_id(
|
|
82
|
+
source: UUID, target: UUID, relation: GraphRelationType, scope_key: str | None,
|
|
83
|
+
) -> UUID:
|
|
84
|
+
return uuid5(GRAPH_NAMESPACE, f"edge:{source}:{target}:{relation.value}:{scope_key or ''}")
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _safe_node_label(label: str | None) -> str | None:
|
|
88
|
+
if not label or label.startswith("memory:"):
|
|
89
|
+
return None
|
|
90
|
+
clean = re.sub(r"\x1b(?:\[[0-?]*[ -/]*[@-~]|\][^\x1b\x07]*(?:\x07|\x1b\\)|.)", "", label)
|
|
91
|
+
clean = "".join(c for c in clean if c.isprintable() and not (0x202A <= ord(c) <= 0x2069))[:120].strip()
|
|
92
|
+
if re.search(
|
|
93
|
+
r"(?i)(?:[a-z]:[\\/]|\\\\|(?:^|\s)/(?:[^/\s]+/)+|"
|
|
94
|
+
r"[a-z][a-z0-9+.-]*://|\b(?:api[_ -]?key|password|passwd|secret|token|authorization)\s*[:=]|"
|
|
95
|
+
r"\bbearer\s+\S+|\bsk-(?:proj-|ant-)?[a-z0-9_-]{8,}\b)",
|
|
96
|
+
clean,
|
|
97
|
+
):
|
|
98
|
+
return None
|
|
99
|
+
return clean or None
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
@dataclass(frozen=True)
|
|
104
|
+
class ExtractedEntity:
|
|
105
|
+
label: str
|
|
106
|
+
key: str
|
|
107
|
+
node_type: GraphNodeType
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@dataclass(frozen=True)
|
|
111
|
+
class ExtractedRelation:
|
|
112
|
+
source: ExtractedEntity
|
|
113
|
+
target: ExtractedEntity
|
|
114
|
+
relation_type: GraphRelationType
|
|
115
|
+
scope_key: str | None
|
|
116
|
+
confidence: float
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class DeterministicEntityExtractor:
|
|
120
|
+
"""Extract only named entities and explicitly worded relations."""
|
|
121
|
+
|
|
122
|
+
def entities(self, text: str) -> list[ExtractedEntity]:
|
|
123
|
+
found: dict[tuple[GraphNodeType, str], ExtractedEntity] = {}
|
|
124
|
+
for relation in self.relations(text):
|
|
125
|
+
for entity in (relation.source, relation.target):
|
|
126
|
+
found[(entity.node_type, entity.key)] = entity
|
|
127
|
+
for match in re.finditer(r"\bProject\s+([A-Za-z][A-Za-z0-9_-]*)", text, re.I):
|
|
128
|
+
entity = self._entity(match.group(1), GraphNodeType.PROJECT)
|
|
129
|
+
found[(entity.node_type, entity.key)] = entity
|
|
130
|
+
for match in re.finditer(r"[A-Za-z][A-Za-z0-9+#._-]*", text):
|
|
131
|
+
label = match.group(0)
|
|
132
|
+
key = canonical_key(label)
|
|
133
|
+
if key in _KNOWN_TOOLS:
|
|
134
|
+
entity = self._entity(label, GraphNodeType.TOOL)
|
|
135
|
+
found[(entity.node_type, entity.key)] = entity
|
|
136
|
+
return sorted(found.values(), key=lambda item: (item.node_type.value, item.key))
|
|
137
|
+
|
|
138
|
+
def query_entities(self, text: str) -> list[ExtractedEntity]:
|
|
139
|
+
values = self.entities(text)
|
|
140
|
+
seen = {(item.node_type, item.key) for item in values}
|
|
141
|
+
for match in re.finditer(r"\b[A-Z][A-Za-z0-9+#._-]+\b", text):
|
|
142
|
+
label = match.group(0)
|
|
143
|
+
key = canonical_key(label)
|
|
144
|
+
if key not in _STOP_LABELS:
|
|
145
|
+
node_type = GraphNodeType.TOOL if key in _KNOWN_TOOLS else GraphNodeType.PROJECT
|
|
146
|
+
item = self._entity(label, node_type)
|
|
147
|
+
if (item.node_type, item.key) not in seen:
|
|
148
|
+
values.append(item)
|
|
149
|
+
seen.add((item.node_type, item.key))
|
|
150
|
+
return values
|
|
151
|
+
|
|
152
|
+
def relations(self, text: str, *, negated: bool = False) -> list[ExtractedRelation]:
|
|
153
|
+
if negated or _NEGATED.search(text):
|
|
154
|
+
return []
|
|
155
|
+
output: list[ExtractedRelation] = []
|
|
156
|
+
# A conjunction beginning another explicit subject is two independent
|
|
157
|
+
# claims, never one long object phrase. Keep this narrow so ordinary
|
|
158
|
+
# tool names and prose cannot manufacture extra relations.
|
|
159
|
+
clauses = re.split(
|
|
160
|
+
rf"\s+(?:and|;)+\s+(?=(?:Project\s+)?{_TOKEN}\s+(?:uses?|used|runs?|depends\s+on|is\s+part\s+of|belongs\s+to|works\s+on)\b)",
|
|
161
|
+
text,
|
|
162
|
+
flags=re.I,
|
|
163
|
+
)
|
|
164
|
+
if len(clauses) > 1:
|
|
165
|
+
for clause in clauses:
|
|
166
|
+
output.extend(self.relations(clause, negated=negated))
|
|
167
|
+
return output
|
|
168
|
+
replacement = _CURRENT_REPLACEMENT.search(text)
|
|
169
|
+
if replacement:
|
|
170
|
+
source = self._entity(replacement.group("source"), GraphNodeType.PROJECT)
|
|
171
|
+
target = self._entity(replacement.group("target"), GraphNodeType.TOOL)
|
|
172
|
+
return [ExtractedRelation(
|
|
173
|
+
source=source, target=target, relation_type=GraphRelationType.USES,
|
|
174
|
+
scope_key=source.key, confidence=0.95,
|
|
175
|
+
)]
|
|
176
|
+
for match in _RELATION.finditer(text):
|
|
177
|
+
source_label = match.group("source").strip()
|
|
178
|
+
source_label = re.sub(
|
|
179
|
+
r"\s+(?:currently|now|previously|formerly)$", "", source_label,
|
|
180
|
+
flags=re.I,
|
|
181
|
+
)
|
|
182
|
+
target_label = match.group("target").strip().rstrip(".,;:!?")
|
|
183
|
+
target_label = re.split(
|
|
184
|
+
r"\s+(?:instead\s+of|previously|formerly|because|due\s+to|but\s+now)\b",
|
|
185
|
+
target_label, maxsplit=1, flags=re.I,
|
|
186
|
+
)[0]
|
|
187
|
+
target_label = re.sub(
|
|
188
|
+
r"\s+(?:for\s+local\s+inference|locally|on\s+the\s+local\s+machine)$",
|
|
189
|
+
"", target_label, flags=re.I,
|
|
190
|
+
).strip()
|
|
191
|
+
if not target_label:
|
|
192
|
+
continue
|
|
193
|
+
verb = " ".join(match.group("verb").casefold().split())
|
|
194
|
+
relation_type = _RELATION_MAP.get(verb)
|
|
195
|
+
if relation_type is None:
|
|
196
|
+
continue
|
|
197
|
+
source_key = canonical_key(source_label)
|
|
198
|
+
target_key = canonical_key(target_label)
|
|
199
|
+
source_type = (
|
|
200
|
+
GraphNodeType.TOOL
|
|
201
|
+
if relation_type == GraphRelationType.RUNS or source_key in _KNOWN_TOOLS
|
|
202
|
+
else GraphNodeType.PROJECT
|
|
203
|
+
)
|
|
204
|
+
target_type = (
|
|
205
|
+
GraphNodeType.PROJECT
|
|
206
|
+
if target_label.casefold().startswith("project ")
|
|
207
|
+
else GraphNodeType.TOOL
|
|
208
|
+
)
|
|
209
|
+
source = self._entity(source_label, source_type)
|
|
210
|
+
target = self._entity(target_label, target_type)
|
|
211
|
+
scope = source.key if source.node_type == GraphNodeType.PROJECT else None
|
|
212
|
+
output.append(ExtractedRelation(
|
|
213
|
+
source=source, target=target, relation_type=relation_type,
|
|
214
|
+
scope_key=scope, confidence=0.95,
|
|
215
|
+
))
|
|
216
|
+
return output
|
|
217
|
+
|
|
218
|
+
@staticmethod
|
|
219
|
+
def _entity(label: str, node_type: GraphNodeType) -> ExtractedEntity:
|
|
220
|
+
clean = re.sub(r"^project\s+", "", label.strip(), flags=re.I).rstrip(".,;:!?")
|
|
221
|
+
return ExtractedEntity(label=clean, key=canonical_key(clean), node_type=node_type)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
class MemoryGraphService:
|
|
225
|
+
"""Build and query a conservative graph derived from authoritative memories."""
|
|
226
|
+
|
|
227
|
+
_projected_statuses = (
|
|
228
|
+
MemoryStatus.ACTIVE, MemoryStatus.HISTORICAL, MemoryStatus.SUPERSEDED,
|
|
229
|
+
MemoryStatus.CONTRADICTED, MemoryStatus.EXPIRED,
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
def __init__(
|
|
233
|
+
self,
|
|
234
|
+
*,
|
|
235
|
+
memory_repo: MemoryRepository,
|
|
236
|
+
relation_repo: RelationRepository,
|
|
237
|
+
graph_repo: SqliteGraphRepository,
|
|
238
|
+
extractor: DeterministicEntityExtractor | None = None,
|
|
239
|
+
) -> None:
|
|
240
|
+
self._memory_repo = memory_repo
|
|
241
|
+
self._relation_repo = relation_repo
|
|
242
|
+
self._graph_repo = graph_repo
|
|
243
|
+
self._extractor = extractor or DeterministicEntityExtractor()
|
|
244
|
+
self._lock = asyncio.Lock()
|
|
245
|
+
|
|
246
|
+
async def ensure_current(self, *, force: bool = False) -> bool:
|
|
247
|
+
async with self._lock:
|
|
248
|
+
if not force and not await self._graph_repo.source_is_dirty():
|
|
249
|
+
return False
|
|
250
|
+
memories = await self._memories()
|
|
251
|
+
await self._rebuild_from(memories)
|
|
252
|
+
return True
|
|
253
|
+
|
|
254
|
+
async def rebuild(self) -> tuple[int, int, int]:
|
|
255
|
+
await self.ensure_current(force=True)
|
|
256
|
+
return await self._graph_repo.counts()
|
|
257
|
+
|
|
258
|
+
async def upsert_memory(self, memory: Memory) -> tuple[int, int, int]:
|
|
259
|
+
del memory
|
|
260
|
+
return await self.rebuild()
|
|
261
|
+
|
|
262
|
+
async def remove_memory(self, memory_id: UUID) -> tuple[int, int, int]:
|
|
263
|
+
del memory_id
|
|
264
|
+
return await self.rebuild()
|
|
265
|
+
|
|
266
|
+
async def find_entities(self, text: str) -> list[GraphNode]:
|
|
267
|
+
await self.ensure_current()
|
|
268
|
+
keys = {item.key for item in self._extractor.query_entities(text)}
|
|
269
|
+
return await self._graph_repo.find_nodes(keys)
|
|
270
|
+
|
|
271
|
+
async def neighbors(
|
|
272
|
+
self,
|
|
273
|
+
node_id: UUID,
|
|
274
|
+
*,
|
|
275
|
+
relation_types: set[GraphRelationType] | None = None,
|
|
276
|
+
min_confidence: float = 0.0,
|
|
277
|
+
) -> list[GraphEdge]:
|
|
278
|
+
await self.ensure_current()
|
|
279
|
+
edges = await self._graph_repo.edges_for_nodes({node_id})
|
|
280
|
+
return [
|
|
281
|
+
edge for edge in edges
|
|
282
|
+
if edge.confidence >= min_confidence
|
|
283
|
+
and (relation_types is None or edge.relation_type in relation_types)
|
|
284
|
+
]
|
|
285
|
+
|
|
286
|
+
async def expand(
|
|
287
|
+
self,
|
|
288
|
+
*,
|
|
289
|
+
query_text: str,
|
|
290
|
+
seed_memory_ids: list[UUID] | None = None,
|
|
291
|
+
max_hops: int = 2,
|
|
292
|
+
min_confidence: float = 0.6,
|
|
293
|
+
max_nodes: int = 100,
|
|
294
|
+
max_edges: int = 250,
|
|
295
|
+
relation_types: set[GraphRelationType] | None = None,
|
|
296
|
+
) -> GraphExpansion:
|
|
297
|
+
if not 1 <= max_hops <= 3:
|
|
298
|
+
raise ValueError("max_hops must be between 1 and 3")
|
|
299
|
+
await self.ensure_current()
|
|
300
|
+
query_nodes = await self.find_entities(query_text)
|
|
301
|
+
seed_ids = {node.id for node in query_nodes}
|
|
302
|
+
for memory_id in seed_memory_ids or []:
|
|
303
|
+
seed_ids.add(stable_node_id(GraphNodeType.MEMORY, str(memory_id)))
|
|
304
|
+
if not seed_ids:
|
|
305
|
+
return GraphExpansion()
|
|
306
|
+
|
|
307
|
+
all_nodes: dict[UUID, GraphNode] = {}
|
|
308
|
+
for seed in seed_ids:
|
|
309
|
+
node = await self._graph_repo.get_node(seed)
|
|
310
|
+
if node is not None:
|
|
311
|
+
all_nodes[seed] = node
|
|
312
|
+
|
|
313
|
+
candidate_scores: dict[UUID, float] = {}
|
|
314
|
+
candidate_paths: dict[UUID, list[GraphPath]] = defaultdict(list)
|
|
315
|
+
visited: set[UUID] = set(seed_ids)
|
|
316
|
+
traversed: set[UUID] = set()
|
|
317
|
+
queue = deque(
|
|
318
|
+
(seed, seed, [seed], [], 1.0, self._seed_scope(all_nodes.get(seed)))
|
|
319
|
+
for seed in sorted(seed_ids, key=str)
|
|
320
|
+
)
|
|
321
|
+
while queue and len(visited) <= max_nodes and len(traversed) < max_edges:
|
|
322
|
+
seed, current, node_path, edge_path, strength, scope = queue.popleft()
|
|
323
|
+
hop = len(edge_path)
|
|
324
|
+
if hop >= max_hops:
|
|
325
|
+
continue
|
|
326
|
+
incident = await self._graph_repo.edges_for_nodes(
|
|
327
|
+
{current}, limit=max_edges - len(traversed)
|
|
328
|
+
)
|
|
329
|
+
adjacent: list[tuple[GraphEdge, UUID]] = []
|
|
330
|
+
for edge in incident:
|
|
331
|
+
if edge.confidence < min_confidence:
|
|
332
|
+
continue
|
|
333
|
+
if relation_types is not None and edge.relation_type not in relation_types:
|
|
334
|
+
continue
|
|
335
|
+
if edge.source_node_id == current:
|
|
336
|
+
adjacent.append((edge, edge.target_node_id))
|
|
337
|
+
elif not edge.directed or edge.relation_type in {
|
|
338
|
+
GraphRelationType.ABOUT, GraphRelationType.MENTIONS,
|
|
339
|
+
}:
|
|
340
|
+
adjacent.append((edge, edge.source_node_id))
|
|
341
|
+
degree = max(1, len(adjacent))
|
|
342
|
+
for edge, neighbor in adjacent:
|
|
343
|
+
if len(traversed) >= max_edges:
|
|
344
|
+
break
|
|
345
|
+
next_scope = scope
|
|
346
|
+
if edge.scope_key:
|
|
347
|
+
if scope and edge.scope_key != scope:
|
|
348
|
+
continue
|
|
349
|
+
next_scope = scope or edge.scope_key
|
|
350
|
+
next_hop = hop + 1
|
|
351
|
+
attenuation = edge.confidence * (0.72 ** next_hop) / (degree ** 0.5)
|
|
352
|
+
contribution = min(1.0, strength * attenuation)
|
|
353
|
+
new_nodes = [*node_path, neighbor]
|
|
354
|
+
new_edges = [*edge_path, edge]
|
|
355
|
+
traversed.add(edge.id)
|
|
356
|
+
if neighbor not in all_nodes:
|
|
357
|
+
neighbor_node = await self._graph_repo.get_node(neighbor)
|
|
358
|
+
if neighbor_node is not None:
|
|
359
|
+
all_nodes[neighbor] = neighbor_node
|
|
360
|
+
support_ids = sorted({item.memory_id for item in edge.supports}, key=str)
|
|
361
|
+
path_nodes: list[GraphPathNode] = []
|
|
362
|
+
for n_id in new_nodes:
|
|
363
|
+
n_obj = all_nodes.get(n_id)
|
|
364
|
+
if n_obj is not None:
|
|
365
|
+
label = _safe_node_label(n_obj.label)
|
|
366
|
+
project_scope = (
|
|
367
|
+
_safe_node_label(n_obj.canonical_key)
|
|
368
|
+
if n_obj.node_type == GraphNodeType.PROJECT else None
|
|
369
|
+
)
|
|
370
|
+
path_nodes.append(GraphPathNode(
|
|
371
|
+
node_id=n_obj.id,
|
|
372
|
+
node_type=n_obj.node_type,
|
|
373
|
+
label=label,
|
|
374
|
+
project_scope=project_scope,
|
|
375
|
+
))
|
|
376
|
+
else:
|
|
377
|
+
path_nodes.append(GraphPathNode(
|
|
378
|
+
node_id=n_id,
|
|
379
|
+
node_type=GraphNodeType.MEMORY,
|
|
380
|
+
label=None,
|
|
381
|
+
project_scope=None,
|
|
382
|
+
))
|
|
383
|
+
|
|
384
|
+
path_edges: list[GraphPathEdge] = []
|
|
385
|
+
for e_obj in new_edges:
|
|
386
|
+
path_edges.append(GraphPathEdge(
|
|
387
|
+
edge_type=e_obj.relation_type,
|
|
388
|
+
confidence=e_obj.confidence,
|
|
389
|
+
supporting_memory_ids=sorted({s.memory_id for s in e_obj.supports}, key=str),
|
|
390
|
+
project_scope=_safe_node_label(e_obj.scope_key),
|
|
391
|
+
))
|
|
392
|
+
|
|
393
|
+
scope_participated = any(e.project_scope is not None for e in path_edges) or any(
|
|
394
|
+
n.project_scope is not None for n in path_nodes
|
|
395
|
+
)
|
|
396
|
+
scope_match = None
|
|
397
|
+
if scope_participated and scope is not None:
|
|
398
|
+
scope_match = any(e.project_scope == scope for e in path_edges) or any(
|
|
399
|
+
n.project_scope == scope for n in path_nodes
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
for memory_id in support_ids:
|
|
403
|
+
candidate_scores[memory_id] = max(candidate_scores.get(memory_id, 0.0), contribution)
|
|
404
|
+
candidate_paths[memory_id].append(GraphPath(
|
|
405
|
+
seed_node_ids=[seed],
|
|
406
|
+
node_ids=new_nodes,
|
|
407
|
+
node_types=[all_nodes[node].node_type for node in new_nodes if node in all_nodes],
|
|
408
|
+
edge_ids=[item.id for item in new_edges],
|
|
409
|
+
edge_types=[item.relation_type for item in new_edges],
|
|
410
|
+
hop_count=next_hop,
|
|
411
|
+
graph_contribution=contribution,
|
|
412
|
+
source_memory_ids=support_ids,
|
|
413
|
+
path_nodes=path_nodes,
|
|
414
|
+
path_edges=path_edges,
|
|
415
|
+
scope_match=scope_match,
|
|
416
|
+
))
|
|
417
|
+
if neighbor not in visited and len(visited) < max_nodes:
|
|
418
|
+
visited.add(neighbor)
|
|
419
|
+
queue.append((seed, neighbor, new_nodes, new_edges, contribution, next_scope))
|
|
420
|
+
|
|
421
|
+
return GraphExpansion(
|
|
422
|
+
seed_node_ids=sorted(seed_ids, key=str), candidate_scores=candidate_scores,
|
|
423
|
+
candidate_paths={key: value[:3] for key, value in candidate_paths.items()},
|
|
424
|
+
visited_node_ids=sorted(visited, key=str),
|
|
425
|
+
traversed_edge_ids=sorted(traversed, key=str),
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
async def _memories(self) -> list[Memory]:
|
|
429
|
+
values: list[Memory] = []
|
|
430
|
+
for status in self._projected_statuses:
|
|
431
|
+
offset = 0
|
|
432
|
+
while True:
|
|
433
|
+
page = await self._memory_repo.list(MemoryFilters(status=status, limit=500, offset=offset))
|
|
434
|
+
values.extend(page)
|
|
435
|
+
if len(page) < 500:
|
|
436
|
+
break
|
|
437
|
+
offset += len(page)
|
|
438
|
+
return values
|
|
439
|
+
|
|
440
|
+
async def _rebuild_from(self, memories: list[Memory]) -> None:
|
|
441
|
+
now = datetime.now(timezone.utc)
|
|
442
|
+
nodes: dict[UUID, GraphNode] = {}
|
|
443
|
+
edge_data: dict[tuple[UUID, UUID, GraphRelationType, str], GraphEdge] = {}
|
|
444
|
+
|
|
445
|
+
def add_node(node_type: GraphNodeType, key: str, label: str, metadata: dict | None = None) -> UUID:
|
|
446
|
+
identifier = stable_node_id(node_type, key)
|
|
447
|
+
nodes.setdefault(identifier, GraphNode(
|
|
448
|
+
id=identifier, node_type=node_type, canonical_key=key, label=label,
|
|
449
|
+
metadata=metadata or {}, created_at=now, updated_at=now,
|
|
450
|
+
))
|
|
451
|
+
return identifier
|
|
452
|
+
|
|
453
|
+
def add_edge(
|
|
454
|
+
source: UUID, target: UUID, relation: GraphRelationType, memory: Memory,
|
|
455
|
+
confidence: float, scope: str | None = None,
|
|
456
|
+
) -> None:
|
|
457
|
+
key = (source, target, relation, scope or "")
|
|
458
|
+
support = GraphEdgeSupport(
|
|
459
|
+
edge_id=stable_edge_id(source, target, relation, scope), memory_id=memory.id,
|
|
460
|
+
confidence=confidence, provenance_event_id=memory.provenance_event_id,
|
|
461
|
+
created_at=now,
|
|
462
|
+
)
|
|
463
|
+
if key in edge_data:
|
|
464
|
+
existing = edge_data[key]
|
|
465
|
+
if all(item.memory_id != memory.id for item in existing.supports):
|
|
466
|
+
existing.supports.append(support)
|
|
467
|
+
existing.confidence = max(existing.confidence, confidence)
|
|
468
|
+
return
|
|
469
|
+
edge_data[key] = GraphEdge(
|
|
470
|
+
id=support.edge_id, source_node_id=source, target_node_id=target,
|
|
471
|
+
relation_type=relation, confidence=confidence,
|
|
472
|
+
directed=relation not in {
|
|
473
|
+
GraphRelationType.MENTIONS,
|
|
474
|
+
GraphRelationType.ABOUT,
|
|
475
|
+
GraphRelationType.COEXISTS_WITH,
|
|
476
|
+
GraphRelationType.CONTRADICTS,
|
|
477
|
+
},
|
|
478
|
+
scope_key=scope, supports=[support], created_at=now, updated_at=now,
|
|
479
|
+
)
|
|
480
|
+
|
|
481
|
+
by_id = {memory.id: memory for memory in memories}
|
|
482
|
+
for memory in memories:
|
|
483
|
+
entities = self._extractor.entities(memory.content)
|
|
484
|
+
relations = self._extractor.relations(memory.content, negated=memory.negated)
|
|
485
|
+
if not entities:
|
|
486
|
+
continue
|
|
487
|
+
memory_node = add_node(
|
|
488
|
+
GraphNodeType.MEMORY, str(memory.id), f"memory:{memory.id}",
|
|
489
|
+
{"status": memory.status.value},
|
|
490
|
+
)
|
|
491
|
+
primary_keys = {relation.source.key for relation in relations}
|
|
492
|
+
for entity in entities:
|
|
493
|
+
entity_node = add_node(entity.node_type, entity.key, entity.label)
|
|
494
|
+
relation = (
|
|
495
|
+
GraphRelationType.ABOUT if entity.key in primary_keys else GraphRelationType.MENTIONS
|
|
496
|
+
)
|
|
497
|
+
add_edge(memory_node, entity_node, relation, memory, 0.9)
|
|
498
|
+
for relation in relations:
|
|
499
|
+
source = add_node(relation.source.node_type, relation.source.key, relation.source.label)
|
|
500
|
+
target = add_node(relation.target.node_type, relation.target.key, relation.target.label)
|
|
501
|
+
add_edge(
|
|
502
|
+
source, target, relation.relation_type, memory,
|
|
503
|
+
relation.confidence, relation.scope_key,
|
|
504
|
+
)
|
|
505
|
+
|
|
506
|
+
seen_relations: set[UUID] = set()
|
|
507
|
+
mapping = {item.value: GraphRelationType(item.value) for item in GraphRelationType}
|
|
508
|
+
for memory in memories:
|
|
509
|
+
for relation in await self._relation_repo.get_relations(memory.id, "outgoing"):
|
|
510
|
+
if relation.id in seen_relations or relation.target_memory_id not in by_id:
|
|
511
|
+
continue
|
|
512
|
+
seen_relations.add(relation.id)
|
|
513
|
+
graph_type = mapping.get(relation.relation_type.value)
|
|
514
|
+
if graph_type is None:
|
|
515
|
+
continue
|
|
516
|
+
source = add_node(GraphNodeType.MEMORY, str(relation.source_memory_id), f"memory:{relation.source_memory_id}")
|
|
517
|
+
target = add_node(GraphNodeType.MEMORY, str(relation.target_memory_id), f"memory:{relation.target_memory_id}")
|
|
518
|
+
add_edge(source, target, graph_type, memory, relation.confidence)
|
|
519
|
+
|
|
520
|
+
await self._graph_repo.replace_all(list(nodes.values()), list(edge_data.values()))
|
|
521
|
+
|
|
522
|
+
@staticmethod
|
|
523
|
+
def _seed_scope(node: GraphNode | None) -> str | None:
|
|
524
|
+
return node.canonical_key if node and node.node_type == GraphNodeType.PROJECT else None
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""Graph-only and rank-fused graph-augmented memory retrieval."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import time
|
|
6
|
+
|
|
7
|
+
from contextos.core.enums import RetrievalMode
|
|
8
|
+
from contextos.core.models import (
|
|
9
|
+
RetrievalConfig,
|
|
10
|
+
RetrievalQuery,
|
|
11
|
+
RetrievalResult,
|
|
12
|
+
RetrievalTrace,
|
|
13
|
+
ScoredMemory,
|
|
14
|
+
StageTrace,
|
|
15
|
+
)
|
|
16
|
+
from contextos.core.protocols import MemoryRepository
|
|
17
|
+
from contextos.services.graph import MemoryGraphService
|
|
18
|
+
from contextos.services.retrieval import HybridRetrievalEngine
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class GraphAugmentedRetrievalEngine:
|
|
22
|
+
"""Add graph candidates using rank fusion, never raw-score addition."""
|
|
23
|
+
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
*,
|
|
27
|
+
base_engine: HybridRetrievalEngine,
|
|
28
|
+
graph_service: MemoryGraphService,
|
|
29
|
+
memory_repo: MemoryRepository,
|
|
30
|
+
rrf_k: int = 60,
|
|
31
|
+
) -> None:
|
|
32
|
+
self._base = base_engine
|
|
33
|
+
self._graph = graph_service
|
|
34
|
+
self._memory_repo = memory_repo
|
|
35
|
+
self._rrf_k = rrf_k
|
|
36
|
+
|
|
37
|
+
async def retrieve(
|
|
38
|
+
self,
|
|
39
|
+
query: str | RetrievalQuery,
|
|
40
|
+
config: RetrievalConfig | None = None,
|
|
41
|
+
) -> RetrievalResult:
|
|
42
|
+
request = HybridRetrievalEngine._coerce_query(query, config)
|
|
43
|
+
if request.mode not in {RetrievalMode.GRAPH, RetrievalMode.HYBRID_GRAPH}:
|
|
44
|
+
return await self._base.retrieve(request, config)
|
|
45
|
+
|
|
46
|
+
started = time.perf_counter()
|
|
47
|
+
base_result = RetrievalResult(query=request.text)
|
|
48
|
+
if request.mode == RetrievalMode.HYBRID_GRAPH:
|
|
49
|
+
base_request = request.model_copy(update={
|
|
50
|
+
"mode": RetrievalMode.HYBRID,
|
|
51
|
+
"k": min(200, max(request.k * 3, 20)),
|
|
52
|
+
})
|
|
53
|
+
base_result = await self._base.retrieve(base_request, config)
|
|
54
|
+
|
|
55
|
+
graph_started = time.perf_counter()
|
|
56
|
+
expansion = await self._graph.expand(
|
|
57
|
+
query_text=request.text,
|
|
58
|
+
seed_memory_ids=[item.memory.id for item in base_result.memories[:10]],
|
|
59
|
+
max_hops=request.graph_max_hops,
|
|
60
|
+
min_confidence=request.graph_min_confidence,
|
|
61
|
+
max_nodes=request.graph_max_nodes,
|
|
62
|
+
max_edges=request.graph_max_edges,
|
|
63
|
+
)
|
|
64
|
+
graph_ranked = sorted(
|
|
65
|
+
expansion.candidate_scores,
|
|
66
|
+
key=lambda item: (-expansion.candidate_scores[item], str(item)),
|
|
67
|
+
)
|
|
68
|
+
graph_ranks = {memory_id: rank for rank, memory_id in enumerate(graph_ranked, 1)}
|
|
69
|
+
base_ranks = {item.memory.id: rank for rank, item in enumerate(base_result.memories, 1)}
|
|
70
|
+
base_by_id = {item.memory.id: item for item in base_result.memories}
|
|
71
|
+
|
|
72
|
+
identifiers = set(graph_ranks)
|
|
73
|
+
if request.mode == RetrievalMode.HYBRID_GRAPH:
|
|
74
|
+
identifiers.update(base_ranks)
|
|
75
|
+
fused: list[ScoredMemory] = []
|
|
76
|
+
for memory_id in identifiers:
|
|
77
|
+
memory = (
|
|
78
|
+
base_by_id[memory_id].memory
|
|
79
|
+
if memory_id in base_by_id
|
|
80
|
+
else await self._memory_repo.get(memory_id)
|
|
81
|
+
)
|
|
82
|
+
if memory is None or not HybridRetrievalEngine._eligible(memory, request):
|
|
83
|
+
continue
|
|
84
|
+
score = 0.0
|
|
85
|
+
sources: list[str] = []
|
|
86
|
+
if memory_id in base_ranks:
|
|
87
|
+
score += 1.0 / (self._rrf_k + base_ranks[memory_id])
|
|
88
|
+
sources.extend(base_by_id[memory_id].retrieval_sources)
|
|
89
|
+
if memory_id in graph_ranks:
|
|
90
|
+
score += 1.0 / (self._rrf_k + graph_ranks[memory_id])
|
|
91
|
+
sources.append("graph")
|
|
92
|
+
original = base_by_id.get(memory_id)
|
|
93
|
+
fused.append(ScoredMemory(
|
|
94
|
+
memory=memory,
|
|
95
|
+
final_score=score,
|
|
96
|
+
vector_score=original.vector_score if original else None,
|
|
97
|
+
bm25_score=original.bm25_score if original else None,
|
|
98
|
+
lexical_rank=original.lexical_rank if original else None,
|
|
99
|
+
dense_rank=original.dense_rank if original else None,
|
|
100
|
+
rrf_rank=base_ranks.get(memory_id, 0),
|
|
101
|
+
metadata_adjustment=original.metadata_adjustment if original else 0.0,
|
|
102
|
+
retrieval_sources=list(dict.fromkeys(sources)),
|
|
103
|
+
graph_score=expansion.candidate_scores.get(memory_id),
|
|
104
|
+
graph_rank=graph_ranks.get(memory_id),
|
|
105
|
+
graph_paths=expansion.candidate_paths.get(memory_id, []),
|
|
106
|
+
))
|
|
107
|
+
fused.sort(key=lambda item: (-item.final_score, str(item.memory.id)))
|
|
108
|
+
pre_limit_ids = [str(item.memory.id) for item in fused]
|
|
109
|
+
fused = fused[:request.k]
|
|
110
|
+
for rank, item in enumerate(fused, 1):
|
|
111
|
+
item.rank = rank
|
|
112
|
+
|
|
113
|
+
graph_stage = StageTrace(
|
|
114
|
+
stage_name="graph_expansion",
|
|
115
|
+
input_count=len(expansion.seed_node_ids),
|
|
116
|
+
output_count=len(graph_ranked),
|
|
117
|
+
latency_ms=(time.perf_counter() - graph_started) * 1000,
|
|
118
|
+
metadata={
|
|
119
|
+
"seed_node_ids": [str(value) for value in expansion.seed_node_ids],
|
|
120
|
+
"visited_node_ids": [str(value) for value in expansion.visited_node_ids],
|
|
121
|
+
"traversed_edge_ids": [str(value) for value in expansion.traversed_edge_ids],
|
|
122
|
+
"max_hops": request.graph_max_hops,
|
|
123
|
+
"min_confidence": request.graph_min_confidence,
|
|
124
|
+
"fusion": "reciprocal_rank",
|
|
125
|
+
},
|
|
126
|
+
)
|
|
127
|
+
stages = [*base_result.trace.stages, graph_stage] if request.include_trace else []
|
|
128
|
+
trace = RetrievalTrace(
|
|
129
|
+
stages=stages,
|
|
130
|
+
total_latency_ms=(time.perf_counter() - started) * 1000,
|
|
131
|
+
total_candidates=len(identifiers),
|
|
132
|
+
total_results=len(fused),
|
|
133
|
+
lexical_candidate_ids=base_result.trace.lexical_candidate_ids,
|
|
134
|
+
dense_candidate_ids=base_result.trace.dense_candidate_ids,
|
|
135
|
+
pre_limit_candidate_ids=pre_limit_ids[:200],
|
|
136
|
+
channel_candidates_truncated=base_result.trace.channel_candidates_truncated,
|
|
137
|
+
pre_limit_candidates_truncated=len(pre_limit_ids) > 200,
|
|
138
|
+
)
|
|
139
|
+
strategies = dict(base_result.strategy_results)
|
|
140
|
+
strategies["graph"] = [item for item in fused if "graph" in item.retrieval_sources]
|
|
141
|
+
return RetrievalResult(
|
|
142
|
+
query=request.text, memories=fused, strategy_results=strategies, trace=trace,
|
|
143
|
+
)
|