contextos-memory-runtime 1.0.0rc2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. contextos/__init__.py +3 -0
  2. contextos/__main__.py +6 -0
  3. contextos/api/__init__.py +1 -0
  4. contextos/api/routes/__init__.py +1 -0
  5. contextos/api/routes/desktop.py +322 -0
  6. contextos/api/routes/ingest.py +17 -0
  7. contextos/api/routes/memories.py +84 -0
  8. contextos/api/routes/models.py +81 -0
  9. contextos/api/routes/retrieval.py +89 -0
  10. contextos/api/routes/system.py +216 -0
  11. contextos/api/server.py +195 -0
  12. contextos/benchmarks/__init__.py +1 -0
  13. contextos/benchmarks/compilation.py +245 -0
  14. contextos/benchmarks/connectors.py +423 -0
  15. contextos/benchmarks/explainability.py +103 -0
  16. contextos/benchmarks/final.py +406 -0
  17. contextos/benchmarks/graph.py +310 -0
  18. contextos/benchmarks/graph_adversarial.py +525 -0
  19. contextos/benchmarks/mcp.py +324 -0
  20. contextos/benchmarks/model_routing.py +203 -0
  21. contextos/benchmarks/optimization.py +305 -0
  22. contextos/benchmarks/rescue_integration.py +127 -0
  23. contextos/benchmarks/retrieval.py +266 -0
  24. contextos/benchmarks/temporal.py +377 -0
  25. contextos/benchmarks/temporal_hotpath.py +76 -0
  26. contextos/benchmarks/terminal.py +62 -0
  27. contextos/cli/__init__.py +1 -0
  28. contextos/cli/app.py +932 -0
  29. contextos/cli/dashboard.py +174 -0
  30. contextos/cli/formatters.py +299 -0
  31. contextos/config/__init__.py +1 -0
  32. contextos/config/settings.py +160 -0
  33. contextos/connectors/__init__.py +6 -0
  34. contextos/connectors/fake.py +11 -0
  35. contextos/connectors/json_import.py +125 -0
  36. contextos/connectors/local_files.py +102 -0
  37. contextos/connectors/manager.py +293 -0
  38. contextos/connectors/models.py +62 -0
  39. contextos/connectors/protocols.py +11 -0
  40. contextos/core/__init__.py +103 -0
  41. contextos/core/enums.py +489 -0
  42. contextos/core/exceptions.py +293 -0
  43. contextos/core/models.py +1147 -0
  44. contextos/core/protocols.py +549 -0
  45. contextos/daemon/__init__.py +1 -0
  46. contextos/daemon/manager.py +510 -0
  47. contextos/daemon/state.py +127 -0
  48. contextos/daemon/wiring.py +296 -0
  49. contextos/demo.py +217 -0
  50. contextos/embedding/__init__.py +1 -0
  51. contextos/embedding/deterministic.py +76 -0
  52. contextos/embedding/sentence_transformers.py +80 -0
  53. contextos/mcp/__init__.py +5 -0
  54. contextos/mcp/server.py +269 -0
  55. contextos/providers/__init__.py +13 -0
  56. contextos/providers/fake.py +217 -0
  57. contextos/providers/ollama.py +297 -0
  58. contextos/providers/openai_compatible.py +337 -0
  59. contextos/services/__init__.py +1 -0
  60. contextos/services/compilation.py +535 -0
  61. contextos/services/explainability.py +553 -0
  62. contextos/services/extraction.py +311 -0
  63. contextos/services/graph.py +524 -0
  64. contextos/services/graph_retrieval.py +143 -0
  65. contextos/services/ingestion.py +143 -0
  66. contextos/services/inspection.py +174 -0
  67. contextos/services/memory.py +291 -0
  68. contextos/services/model_service.py +409 -0
  69. contextos/services/optimization.py +426 -0
  70. contextos/services/privacy.py +331 -0
  71. contextos/services/retrieval.py +302 -0
  72. contextos/services/retrieval_index.py +88 -0
  73. contextos/services/router.py +302 -0
  74. contextos/services/secret_scanner.py +207 -0
  75. contextos/services/telemetry_query.py +102 -0
  76. contextos/services/temporal.py +500 -0
  77. contextos/services/token_counter.py +222 -0
  78. contextos/storage/__init__.py +1 -0
  79. contextos/storage/connector_repo.py +67 -0
  80. contextos/storage/database.py +497 -0
  81. contextos/storage/event_repo.py +137 -0
  82. contextos/storage/graph_repo.py +228 -0
  83. contextos/storage/lexical/__init__.py +1 -0
  84. contextos/storage/lexical/bm25.py +134 -0
  85. contextos/storage/memory_repo.py +589 -0
  86. contextos/storage/relation_repo.py +80 -0
  87. contextos/storage/telemetry_repo.py +481 -0
  88. contextos/storage/vector/__init__.py +1 -0
  89. contextos/storage/vector/in_memory.py +162 -0
  90. contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
  91. contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
  92. contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
  93. contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,228 @@
1
+ """SQLite persistence for the rebuildable ContextOS memory graph."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import json
7
+ import sqlite3
8
+ from datetime import datetime, timezone
9
+ from uuid import UUID
10
+
11
+ import aiosqlite
12
+
13
+ from contextos.core.enums import GraphNodeType, GraphRelationType
14
+ from contextos.core.models import GraphEdge, GraphEdgeSupport, GraphNode
15
+
16
+
17
+ def _dt(value: str) -> datetime:
18
+ parsed = datetime.fromisoformat(value)
19
+ return parsed.replace(tzinfo=timezone.utc) if parsed.tzinfo is None else parsed
20
+
21
+
22
+ class SqliteGraphRepository:
23
+ """Persistent graph projection sharing the main SQLite database."""
24
+
25
+ def __init__(self, db: aiosqlite.Connection) -> None:
26
+ self._db = db
27
+
28
+ async def replace_all(
29
+ self,
30
+ nodes: list[GraphNode],
31
+ edges: list[GraphEdge],
32
+ ) -> None:
33
+ """Atomically replace the derived graph; retain the old graph on failure."""
34
+ node_ids = {node.id for node in nodes}
35
+ if len(node_ids) != len(nodes):
36
+ raise ValueError("Duplicate graph node ID")
37
+ for edge in edges:
38
+ if edge.source_node_id not in node_ids or edge.target_node_id not in node_ids:
39
+ raise ValueError("Graph edge references an unknown node")
40
+ if not edge.supports:
41
+ raise ValueError("Graph edges require at least one support")
42
+
43
+ for attempt in range(5):
44
+ try:
45
+ await self._db.execute("BEGIN IMMEDIATE")
46
+ break
47
+ except Exception as exc:
48
+ msg = str(exc).lower()
49
+ if attempt < 4 and ("locked" in msg or "busy" in msg or "cannot start a transaction" in msg):
50
+ await asyncio.sleep(0.01 * (2 ** attempt))
51
+ continue
52
+ raise
53
+ try:
54
+ await self._db.execute("DELETE FROM graph_edge_supports")
55
+ await self._db.execute("DELETE FROM graph_edges")
56
+ await self._db.execute("DELETE FROM graph_nodes")
57
+ await self._db.executemany(
58
+ "INSERT INTO graph_nodes "
59
+ "(id, node_type, canonical_key, label, metadata, created_at, updated_at) "
60
+ "VALUES (?, ?, ?, ?, ?, ?, ?)",
61
+ [
62
+ (
63
+ str(node.id), node.node_type.value, node.canonical_key, node.label,
64
+ json.dumps(node.metadata, sort_keys=True),
65
+ node.created_at.astimezone(timezone.utc).isoformat(),
66
+ node.updated_at.astimezone(timezone.utc).isoformat(),
67
+ )
68
+ for node in nodes
69
+ ],
70
+ )
71
+ await self._db.executemany(
72
+ "INSERT INTO graph_edges "
73
+ "(id, source_node_id, target_node_id, relation_type, confidence, directed, "
74
+ "scope_key, metadata, created_at, updated_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
75
+ [
76
+ (
77
+ str(edge.id), str(edge.source_node_id), str(edge.target_node_id),
78
+ edge.relation_type.value, edge.confidence, int(edge.directed),
79
+ edge.scope_key or "", json.dumps(edge.metadata, sort_keys=True),
80
+ edge.created_at.astimezone(timezone.utc).isoformat(),
81
+ edge.updated_at.astimezone(timezone.utc).isoformat(),
82
+ )
83
+ for edge in edges
84
+ ],
85
+ )
86
+ supports = [support for edge in edges for support in edge.supports]
87
+ await self._db.executemany(
88
+ "INSERT INTO graph_edge_supports "
89
+ "(edge_id, memory_id, confidence, provenance_event_id, created_at) "
90
+ "VALUES (?, ?, ?, ?, ?)",
91
+ [
92
+ (
93
+ str(item.edge_id), str(item.memory_id), item.confidence,
94
+ str(item.provenance_event_id) if item.provenance_event_id else None,
95
+ item.created_at.astimezone(timezone.utc).isoformat(),
96
+ )
97
+ for item in supports
98
+ ],
99
+ )
100
+ await self._db.execute(
101
+ "UPDATE graph_projection_state SET dirty = 0 WHERE singleton = 1"
102
+ )
103
+ await self._db.commit()
104
+ except Exception:
105
+ await self._db.rollback()
106
+ raise
107
+
108
+ async def nodes(self) -> list[GraphNode]:
109
+ cursor = await self._db.execute("SELECT * FROM graph_nodes ORDER BY node_type, canonical_key")
110
+ return [self._node(row) for row in await cursor.fetchall()]
111
+
112
+ async def find_nodes(self, canonical_keys: set[str]) -> list[GraphNode]:
113
+ if not canonical_keys:
114
+ return []
115
+ placeholders = ", ".join("?" for _ in canonical_keys)
116
+ cursor = await self._db.execute(
117
+ f"SELECT * FROM graph_nodes WHERE canonical_key IN ({placeholders}) "
118
+ "ORDER BY node_type, canonical_key",
119
+ tuple(sorted(canonical_keys)),
120
+ )
121
+ return [self._node(row) for row in await cursor.fetchall()]
122
+
123
+ async def get_node(self, node_id: UUID) -> GraphNode | None:
124
+ cursor = await self._db.execute("SELECT * FROM graph_nodes WHERE id = ?", (str(node_id),))
125
+ row = await cursor.fetchone()
126
+ return self._node(row) if row else None
127
+
128
+ async def edges_for_nodes(
129
+ self, node_ids: set[UUID], *, limit: int | None = None
130
+ ) -> list[GraphEdge]:
131
+ if not node_ids or (limit is not None and limit <= 0):
132
+ return []
133
+ raw = tuple(str(value) for value in sorted(node_ids, key=str))
134
+ placeholders = ", ".join("?" for _ in raw)
135
+ sql = (
136
+ f"SELECT * FROM graph_edges WHERE source_node_id IN ({placeholders}) "
137
+ f"OR target_node_id IN ({placeholders}) ORDER BY id"
138
+ )
139
+ params: tuple[object, ...] = raw + raw
140
+ if limit is not None:
141
+ sql += " LIMIT ?"
142
+ params += (limit,)
143
+ cursor = await self._db.execute(sql, params)
144
+ rows = await cursor.fetchall()
145
+ if not rows:
146
+ return []
147
+ edge_ids = tuple(row["id"] for row in rows)
148
+ support_placeholders = ", ".join("?" for _ in edge_ids)
149
+ support_cursor = await self._db.execute(
150
+ f"SELECT * FROM graph_edge_supports WHERE edge_id IN ({support_placeholders}) "
151
+ "ORDER BY edge_id, memory_id",
152
+ edge_ids,
153
+ )
154
+ grouped: dict[str, list[GraphEdgeSupport]] = {}
155
+ for item in await support_cursor.fetchall():
156
+ grouped.setdefault(item["edge_id"], []).append(self._support(item))
157
+ return [self._edge_from_data(row, grouped.get(row["id"], [])) for row in rows]
158
+
159
+ async def all_edges(self) -> list[GraphEdge]:
160
+ cursor = await self._db.execute("SELECT * FROM graph_edges ORDER BY id")
161
+ rows = await cursor.fetchall()
162
+ support_cursor = await self._db.execute(
163
+ "SELECT * FROM graph_edge_supports ORDER BY edge_id, memory_id"
164
+ )
165
+ grouped: dict[str, list[GraphEdgeSupport]] = {}
166
+ for item in await support_cursor.fetchall():
167
+ grouped.setdefault(item["edge_id"], []).append(self._support(item))
168
+ return [self._edge_from_data(row, grouped.get(row["id"], [])) for row in rows]
169
+
170
+ async def counts(self) -> tuple[int, int, int]:
171
+ values = []
172
+ for table in ("graph_nodes", "graph_edges", "graph_edge_supports"):
173
+ cursor = await self._db.execute(f"SELECT COUNT(*) FROM {table}")
174
+ values.append((await cursor.fetchone())[0])
175
+ return values[0], values[1], values[2]
176
+
177
+ async def source_is_dirty(self) -> bool:
178
+ cursor = await self._db.execute(
179
+ "SELECT dirty FROM graph_projection_state WHERE singleton = 1"
180
+ )
181
+ row = await cursor.fetchone()
182
+ return bool(row[0]) if row else True
183
+
184
+ @staticmethod
185
+ def _node(row: aiosqlite.Row) -> GraphNode:
186
+ data = dict(row)
187
+ return GraphNode(
188
+ id=UUID(data["id"]),
189
+ node_type=GraphNodeType(data["node_type"]),
190
+ canonical_key=data["canonical_key"],
191
+ label=data["label"],
192
+ metadata=json.loads(data["metadata"]),
193
+ created_at=_dt(data["created_at"]),
194
+ updated_at=_dt(data["updated_at"]),
195
+ )
196
+
197
+ async def _edge(self, row: aiosqlite.Row) -> GraphEdge:
198
+ data = dict(row)
199
+ cursor = await self._db.execute(
200
+ "SELECT * FROM graph_edge_supports WHERE edge_id = ? ORDER BY memory_id",
201
+ (data["id"],),
202
+ )
203
+ supports = [self._support(item) for item in await cursor.fetchall()]
204
+ return self._edge_from_data(row, supports)
205
+
206
+ @staticmethod
207
+ def _support(item: aiosqlite.Row) -> GraphEdgeSupport:
208
+ return GraphEdgeSupport(
209
+ edge_id=UUID(item["edge_id"]), memory_id=UUID(item["memory_id"]),
210
+ confidence=item["confidence"],
211
+ provenance_event_id=(
212
+ UUID(item["provenance_event_id"]) if item["provenance_event_id"] else None
213
+ ),
214
+ created_at=_dt(item["created_at"]),
215
+ )
216
+
217
+ @staticmethod
218
+ def _edge_from_data(row: aiosqlite.Row, supports: list[GraphEdgeSupport]) -> GraphEdge:
219
+ data = dict(row)
220
+ return GraphEdge(
221
+ id=UUID(data["id"]), source_node_id=UUID(data["source_node_id"]),
222
+ target_node_id=UUID(data["target_node_id"]),
223
+ relation_type=GraphRelationType(data["relation_type"]),
224
+ confidence=data["confidence"], directed=bool(data["directed"]),
225
+ scope_key=data["scope_key"] or None,
226
+ metadata=json.loads(data["metadata"]), supports=supports,
227
+ created_at=_dt(data["created_at"]), updated_at=_dt(data["updated_at"]),
228
+ )
@@ -0,0 +1 @@
1
+ """Lexical search sub-package for ContextOS."""
@@ -0,0 +1,134 @@
1
+ """Deterministic in-memory BM25 lexical index."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ import re
7
+ from collections import Counter
8
+ from typing import Any
9
+
10
+ from contextos.core.models import LexicalResult
11
+
12
+ _STOP_WORDS = {
13
+ "a", "an", "and", "did", "do", "does", "for", "has", "have", "i", "in",
14
+ "is", "it", "of", "on", "the", "to", "user", "was", "what", "when", "which",
15
+ "with",
16
+ }
17
+
18
+
19
+ def tokenize(text: str) -> list[str]:
20
+ """Normalize case and punctuation while retaining technical identifiers."""
21
+ tokens = re.findall(r"[a-z0-9]+(?:[+#._-][a-z0-9]+)*", text.casefold())
22
+ normalized: list[str] = []
23
+ for token in tokens:
24
+ if token in _STOP_WORDS:
25
+ continue
26
+ if token in {"focused", "focuses", "focusing"}:
27
+ token = "focus"
28
+ elif (
29
+ token != "focus"
30
+ and token.endswith("s")
31
+ and len(token) > 4
32
+ and not token.endswith("ss")
33
+ ):
34
+ token = token[:-1]
35
+ normalized.append(token)
36
+ return normalized
37
+
38
+
39
+ class BM25Index:
40
+ """BM25Okapi-style lexical ranking with deterministic ties."""
41
+
42
+ def __init__(self, *, k1: float = 1.5, b: float = 0.75) -> None:
43
+ self._k1 = k1
44
+ self._b = b
45
+ self._documents: dict[str, list[str]] = {}
46
+ self._metadata: dict[str, dict[str, Any]] = {}
47
+
48
+ def contains(self, doc_id: str | Any) -> bool:
49
+ """Check whether a document ID is present in the lexical index."""
50
+ return str(doc_id) in self._documents
51
+
52
+ def __contains__(self, doc_id: str | Any) -> bool:
53
+ return str(doc_id) in self._documents
54
+
55
+ async def index(
56
+ self, doc_id: str, text: str, metadata: dict[str, Any] | None = None
57
+ ) -> None:
58
+ tokens = tokenize(text)
59
+ if tokens:
60
+ self._documents[doc_id] = tokens
61
+ self._metadata[doc_id] = dict(metadata or {})
62
+ else:
63
+ await self.delete(doc_id)
64
+
65
+ async def search(
66
+ self,
67
+ query: str,
68
+ top_k: int = 20,
69
+ filters: dict[str, Any] | None = None,
70
+ ) -> list[LexicalResult]:
71
+ terms = tokenize(query)
72
+ if not terms or not self._documents or top_k <= 0:
73
+ return []
74
+ candidates = [doc_id for doc_id in self._documents if self._matches(doc_id, filters)]
75
+ if not candidates:
76
+ return []
77
+ corpus_size = len(self._documents)
78
+ average_length = sum(map(len, self._documents.values())) / corpus_size
79
+ document_frequency = {
80
+ term: sum(term in tokens for tokens in self._documents.values())
81
+ for term in set(terms)
82
+ }
83
+ ranked: list[tuple[str, float]] = []
84
+ for doc_id in candidates:
85
+ tokens = self._documents[doc_id]
86
+ frequencies = Counter(tokens)
87
+ score = 0.0
88
+ for term in terms:
89
+ frequency = frequencies[term]
90
+ if not frequency:
91
+ continue
92
+ df = document_frequency[term]
93
+ # Positive Robertson/Sparck Jones IDF, as used by Lucene.
94
+ inverse_document_frequency = math.log(
95
+ 1.0 + (corpus_size - df + 0.5) / (df + 0.5)
96
+ )
97
+ denominator = frequency + self._k1 * (
98
+ 1.0 - self._b + self._b * len(tokens) / average_length
99
+ )
100
+ score += inverse_document_frequency * frequency * (self._k1 + 1.0) / denominator
101
+ if score > 0.0:
102
+ ranked.append((doc_id, score))
103
+ ranked.sort(key=lambda item: (-item[1], item[0]))
104
+ return [
105
+ LexicalResult(id=doc_id, score=score, metadata=self._metadata[doc_id])
106
+ for doc_id, score in ranked[:top_k]
107
+ ]
108
+
109
+ async def delete(self, doc_id: str) -> None:
110
+ self._documents.pop(doc_id, None)
111
+ self._metadata.pop(doc_id, None)
112
+
113
+ async def count(self) -> int:
114
+ return len(self._documents)
115
+
116
+ async def rebuild(self, documents: dict[str, str]) -> None:
117
+ self._documents = {
118
+ doc_id: tokens for doc_id, text in documents.items() if (tokens := tokenize(text))
119
+ }
120
+ self._metadata = {doc_id: {} for doc_id in self._documents}
121
+
122
+ async def rebuild_with_metadata(
123
+ self, documents: dict[str, tuple[str, dict[str, Any]]]
124
+ ) -> None:
125
+ self._documents.clear()
126
+ self._metadata.clear()
127
+ for doc_id, (text, metadata) in documents.items():
128
+ await self.index(doc_id, text, metadata)
129
+
130
+ def _matches(self, doc_id: str, filters: dict[str, Any] | None) -> bool:
131
+ if not filters:
132
+ return True
133
+ metadata = self._metadata.get(doc_id, {})
134
+ return all(metadata.get(key) == value for key, value in filters.items())