cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/trace/store.py
ADDED
|
@@ -0,0 +1,680 @@
|
|
|
1
|
+
"""The Symbolic Trace — bi-temporal fact store on SQLite (WAL).
|
|
2
|
+
|
|
3
|
+
Implements the Section 1.1 data model: Subject-Relation-Value triples
|
|
4
|
+
with valid/transaction time, contradiction + temporal + provenance
|
|
5
|
+
edges, source chunks, Memory-Git commits (hash-chained) and branches.
|
|
6
|
+
SQLite is the embedded substrate for local/edge operation; the store is
|
|
7
|
+
designed so an ArcadeDB backend can replace it behind the same API
|
|
8
|
+
(the plan's production graph engine).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import sqlite3
|
|
15
|
+
import threading
|
|
16
|
+
from datetime import datetime, timezone
|
|
17
|
+
|
|
18
|
+
from cortexm.errors import BranchError, StoreError
|
|
19
|
+
from cortexm.security.hashes import HashProvider
|
|
20
|
+
from cortexm.trace.fact import Fact
|
|
21
|
+
from cortexm.util import iso, new_id, parse_ts, token_estimate
|
|
22
|
+
|
|
23
|
+
SCHEMA = """
|
|
24
|
+
CREATE TABLE IF NOT EXISTS facts (
|
|
25
|
+
id TEXT PRIMARY KEY,
|
|
26
|
+
subject TEXT NOT NULL, relation TEXT NOT NULL, value TEXT NOT NULL,
|
|
27
|
+
valid_from TEXT NOT NULL, valid_to TEXT,
|
|
28
|
+
tx_from TEXT NOT NULL, tx_to TEXT,
|
|
29
|
+
confidence REAL DEFAULT 0.8,
|
|
30
|
+
source_hash TEXT DEFAULT '', source_id TEXT DEFAULT '',
|
|
31
|
+
user_id TEXT DEFAULT 'default', agent_id TEXT, run_id TEXT,
|
|
32
|
+
memory_type TEXT DEFAULT 'short_term',
|
|
33
|
+
access_count INTEGER DEFAULT 0, reinforcement INTEGER DEFAULT 1,
|
|
34
|
+
is_active INTEGER DEFAULT 1, is_derived INTEGER DEFAULT 0,
|
|
35
|
+
quarantined INTEGER DEFAULT 0,
|
|
36
|
+
birth_commit TEXT, retired_commit TEXT,
|
|
37
|
+
provenance TEXT DEFAULT '{}'
|
|
38
|
+
);
|
|
39
|
+
CREATE INDEX IF NOT EXISTS idx_facts_sr ON facts(subject, relation);
|
|
40
|
+
CREATE INDEX IF NOT EXISTS idx_facts_user ON facts(user_id);
|
|
41
|
+
CREATE INDEX IF NOT EXISTS idx_facts_active ON facts(is_active);
|
|
42
|
+
CREATE INDEX IF NOT EXISTS idx_facts_rel ON facts(relation);
|
|
43
|
+
CREATE INDEX IF NOT EXISTS idx_facts_valid ON facts(valid_from);
|
|
44
|
+
CREATE INDEX IF NOT EXISTS idx_facts_birth ON facts(birth_commit);
|
|
45
|
+
-- composite indexes added for the v2 SPARQL + REST query paths
|
|
46
|
+
CREATE INDEX IF NOT EXISTS idx_facts_user_active ON facts(user_id, is_active);
|
|
47
|
+
CREATE INDEX IF NOT EXISTS idx_facts_value ON facts(value);
|
|
48
|
+
CREATE INDEX IF NOT EXISTS idx_facts_subject_value ON facts(subject, value);
|
|
49
|
+
|
|
50
|
+
CREATE TABLE IF NOT EXISTS edges (
|
|
51
|
+
src TEXT NOT NULL, dst TEXT NOT NULL, kind TEXT NOT NULL,
|
|
52
|
+
meta TEXT DEFAULT '{}', created TEXT,
|
|
53
|
+
PRIMARY KEY (src, dst, kind)
|
|
54
|
+
);
|
|
55
|
+
CREATE INDEX IF NOT EXISTS idx_edges_src ON edges(src);
|
|
56
|
+
CREATE INDEX IF NOT EXISTS idx_edges_dst ON edges(dst);
|
|
57
|
+
-- kind index — critical for SPARQL `?a edge:CAUSAL ?b` queries that
|
|
58
|
+
-- post-filter on the kind column
|
|
59
|
+
CREATE INDEX IF NOT EXISTS idx_edges_kind ON edges(kind);
|
|
60
|
+
|
|
61
|
+
CREATE TABLE IF NOT EXISTS chunks (
|
|
62
|
+
id TEXT PRIMARY KEY, text TEXT NOT NULL,
|
|
63
|
+
user_id TEXT, agent_id TEXT, run_id TEXT,
|
|
64
|
+
ts TEXT, source TEXT DEFAULT '', hash TEXT DEFAULT '', tokens INTEGER DEFAULT 0
|
|
65
|
+
);
|
|
66
|
+
CREATE INDEX IF NOT EXISTS idx_chunks_user ON chunks(user_id);
|
|
67
|
+
|
|
68
|
+
CREATE TABLE IF NOT EXISTS commits (
|
|
69
|
+
id TEXT PRIMARY KEY, parents TEXT DEFAULT '[]', branch TEXT,
|
|
70
|
+
message TEXT DEFAULT '', ts TEXT, chain_hash TEXT, n_facts INTEGER DEFAULT 0
|
|
71
|
+
);
|
|
72
|
+
CREATE TABLE IF NOT EXISTS branches (
|
|
73
|
+
name TEXT PRIMARY KEY, head TEXT NOT NULL, created TEXT
|
|
74
|
+
);
|
|
75
|
+
CREATE TABLE IF NOT EXISTS kv (k TEXT PRIMARY KEY, v TEXT);
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
FACT_COLUMNS = ("id, subject, relation, value, valid_from, valid_to, tx_from, tx_to, "
|
|
79
|
+
"confidence, source_hash, source_id, user_id, agent_id, run_id, "
|
|
80
|
+
"memory_type, access_count, reinforcement, is_active, is_derived, "
|
|
81
|
+
"quarantined, birth_commit, retired_commit, provenance")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class _SafeCursor:
|
|
85
|
+
"""Cursor-like object over eagerly-materialized rows (thread-safe)."""
|
|
86
|
+
|
|
87
|
+
def __init__(self, rows, rowcount, lastrowid, description) -> None:
|
|
88
|
+
self._rows = rows
|
|
89
|
+
self._iter = iter(rows)
|
|
90
|
+
self.rowcount = rowcount
|
|
91
|
+
self.lastrowid = lastrowid
|
|
92
|
+
self.description = description
|
|
93
|
+
self.arraysize = 1
|
|
94
|
+
|
|
95
|
+
def fetchone(self):
|
|
96
|
+
try:
|
|
97
|
+
return next(self._iter)
|
|
98
|
+
except StopIteration:
|
|
99
|
+
return None
|
|
100
|
+
|
|
101
|
+
def fetchall(self):
|
|
102
|
+
rest = list(self._iter)
|
|
103
|
+
self._iter = iter([])
|
|
104
|
+
return rest
|
|
105
|
+
|
|
106
|
+
def fetchmany(self, size=None):
|
|
107
|
+
out = []
|
|
108
|
+
for _ in range(size or 1):
|
|
109
|
+
try:
|
|
110
|
+
out.append(next(self._iter))
|
|
111
|
+
except StopIteration:
|
|
112
|
+
break
|
|
113
|
+
return out
|
|
114
|
+
|
|
115
|
+
def __iter__(self):
|
|
116
|
+
return self
|
|
117
|
+
|
|
118
|
+
def __next__(self):
|
|
119
|
+
return next(self._iter)
|
|
120
|
+
|
|
121
|
+
def close(self) -> None:
|
|
122
|
+
pass
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class SafeConnection:
|
|
126
|
+
"""Serializes every statement on one SQLite connection.
|
|
127
|
+
|
|
128
|
+
SQLite connections are not safe for concurrent cursor use even with
|
|
129
|
+
``check_same_thread=False`` — interleaved commit/iterate produces
|
|
130
|
+
InterfaceError('bad parameter or other API misuse'). This wrapper
|
|
131
|
+
holds an RLock across execute+materialize and across commit, making
|
|
132
|
+
the whole TraceStore safe under multi-threaded load (REST server,
|
|
133
|
+
concurrent writers). Eager materialization keeps semantics: callers
|
|
134
|
+
only use fetchone/fetchall/iteration/rowcount/lastrowid.
|
|
135
|
+
"""
|
|
136
|
+
|
|
137
|
+
def __init__(self, conn) -> None:
|
|
138
|
+
self._conn = conn
|
|
139
|
+
self._lock = threading.RLock()
|
|
140
|
+
|
|
141
|
+
def execute(self, sql, params=()):
|
|
142
|
+
with self._lock:
|
|
143
|
+
cur = self._conn.execute(sql, params)
|
|
144
|
+
try:
|
|
145
|
+
rows = cur.fetchall()
|
|
146
|
+
except Exception:
|
|
147
|
+
rows = []
|
|
148
|
+
rc, lrid, desc = cur.rowcount, cur.lastrowid, cur.description
|
|
149
|
+
cur.close()
|
|
150
|
+
return _SafeCursor(rows, rc, lrid, desc)
|
|
151
|
+
|
|
152
|
+
def executemany(self, sql, seq):
|
|
153
|
+
with self._lock:
|
|
154
|
+
cur = self._conn.executemany(sql, seq)
|
|
155
|
+
rc = cur.rowcount
|
|
156
|
+
cur.close()
|
|
157
|
+
return _SafeCursor([], rc, None, None)
|
|
158
|
+
|
|
159
|
+
def executescript(self, script):
|
|
160
|
+
with self._lock:
|
|
161
|
+
return self._conn.executescript(script)
|
|
162
|
+
|
|
163
|
+
def commit(self):
|
|
164
|
+
with self._lock:
|
|
165
|
+
self._conn.commit()
|
|
166
|
+
|
|
167
|
+
def rollback(self):
|
|
168
|
+
with self._lock:
|
|
169
|
+
self._conn.rollback()
|
|
170
|
+
|
|
171
|
+
def close(self):
|
|
172
|
+
with self._lock:
|
|
173
|
+
self._conn.close()
|
|
174
|
+
|
|
175
|
+
@property
|
|
176
|
+
def row_factory(self):
|
|
177
|
+
return self._conn.row_factory
|
|
178
|
+
|
|
179
|
+
@row_factory.setter
|
|
180
|
+
def row_factory(self, v):
|
|
181
|
+
with self._lock:
|
|
182
|
+
self._conn.row_factory = v
|
|
183
|
+
|
|
184
|
+
@property
|
|
185
|
+
def in_transaction(self):
|
|
186
|
+
return self._conn.in_transaction
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
class TraceStore:
|
|
190
|
+
def __init__(self, db_path: str = ":memory:", provider: HashProvider | None = None,
|
|
191
|
+
wal_sync: str = "normal") -> None:
|
|
192
|
+
self.db_path = db_path
|
|
193
|
+
self.hasher = provider or HashProvider()
|
|
194
|
+
mem = db_path in (":memory:", None, "")
|
|
195
|
+
self.conn = SafeConnection(
|
|
196
|
+
sqlite3.connect(db_path or ":memory:", check_same_thread=False))
|
|
197
|
+
if not mem:
|
|
198
|
+
# Aeon-inspired crash-recoverable write path:
|
|
199
|
+
# journal_mode=WAL — readers never block the writer, and a
|
|
200
|
+
# torn write rolls back cleanly on reopen.
|
|
201
|
+
# synchronous — NORMAL: commits survive process crash
|
|
202
|
+
# (SIGKILL) at full speed; FULL additionally
|
|
203
|
+
# survives OS/power loss at fsync cost.
|
|
204
|
+
self.conn.execute("PRAGMA journal_mode=WAL")
|
|
205
|
+
self.conn.execute(
|
|
206
|
+
"PRAGMA synchronous="
|
|
207
|
+
+ ("FULL" if str(wal_sync).lower() == "full" else "NORMAL"))
|
|
208
|
+
self.conn.row_factory = sqlite3.Row
|
|
209
|
+
self.conn.executescript(SCHEMA)
|
|
210
|
+
self._ancestry_cache: dict[str, frozenset[str]] = {}
|
|
211
|
+
self._active_cache: dict[str, frozenset[str]] = {}
|
|
212
|
+
self._batching = False
|
|
213
|
+
self._ensure_genesis()
|
|
214
|
+
|
|
215
|
+
def checkpoint(self, mode: str = "TRUNCATE") -> None:
|
|
216
|
+
"""Fold the WAL back into the main db file (shrink + fast reopen)."""
|
|
217
|
+
if self.db_path in (":memory:", None, ""):
|
|
218
|
+
return
|
|
219
|
+
try:
|
|
220
|
+
self.conn.execute(f"PRAGMA wal_checkpoint({mode})")
|
|
221
|
+
except Exception:
|
|
222
|
+
pass
|
|
223
|
+
|
|
224
|
+
def begin_batch(self) -> None:
|
|
225
|
+
self._batching = True
|
|
226
|
+
|
|
227
|
+
def end_batch(self) -> None:
|
|
228
|
+
self._batching = False
|
|
229
|
+
self.conn.commit()
|
|
230
|
+
|
|
231
|
+
def _maybe_commit(self) -> None:
|
|
232
|
+
if not self._batching:
|
|
233
|
+
self.conn.commit()
|
|
234
|
+
|
|
235
|
+
@property
|
|
236
|
+
def batching(self) -> bool:
|
|
237
|
+
return self._batching
|
|
238
|
+
|
|
239
|
+
# ------------------------------------------------------------------ kv
|
|
240
|
+
def kv_get(self, key: str, default: str | None = None) -> str | None:
|
|
241
|
+
row = self.conn.execute("SELECT v FROM kv WHERE k=?", (key,)).fetchone()
|
|
242
|
+
return row["v"] if row else default
|
|
243
|
+
|
|
244
|
+
def kv_set(self, key: str, value: str) -> None:
|
|
245
|
+
self.conn.execute(
|
|
246
|
+
"INSERT INTO kv(k, v) VALUES(?, ?) ON CONFLICT(k) DO UPDATE SET v=excluded.v",
|
|
247
|
+
(key, value))
|
|
248
|
+
self.conn.commit()
|
|
249
|
+
|
|
250
|
+
def iter_kv(self, prefix: str = ""):
|
|
251
|
+
"""Yield (key, value) pairs whose key starts with ``prefix``."""
|
|
252
|
+
cur = self.conn.execute(
|
|
253
|
+
"SELECT k, v FROM kv WHERE k LIKE ? ORDER BY k",
|
|
254
|
+
(prefix + "%",))
|
|
255
|
+
for row in cur:
|
|
256
|
+
yield row["k"], row["v"]
|
|
257
|
+
|
|
258
|
+
def kv_delete(self, key: str) -> None:
|
|
259
|
+
self.conn.execute("DELETE FROM kv WHERE k=?", (key,))
|
|
260
|
+
self.conn.commit()
|
|
261
|
+
|
|
262
|
+
# -------------------------------------------------------------- genesis
|
|
263
|
+
def _ensure_genesis(self) -> None:
|
|
264
|
+
if not self.conn.execute("SELECT 1 FROM branches LIMIT 1").fetchone():
|
|
265
|
+
cid = new_id()
|
|
266
|
+
now = iso(datetime.utcnow().__class__.now() if False else datetime.utcnow()) if False else iso(datetime.now()) # noqa
|
|
267
|
+
chain = self.hasher.hash_text("genesis:" + cid)
|
|
268
|
+
self.conn.execute(
|
|
269
|
+
"INSERT INTO commits(id, parents, branch, message, ts, chain_hash) VALUES(?,?,?,?,?,?)",
|
|
270
|
+
(cid, "[]", "main", "genesis", now, chain))
|
|
271
|
+
self.conn.execute(
|
|
272
|
+
"INSERT INTO branches(name, head, created) VALUES(?,?,?)", ("main", cid, now))
|
|
273
|
+
self.kv_set("HEAD_BRANCH", "main")
|
|
274
|
+
self.kv_set("SCHEMA_VERSION", "1")
|
|
275
|
+
|
|
276
|
+
# -------------------------------------------------------------- chunks
|
|
277
|
+
def add_chunk(self, text: str, *, user_id: str = "default", agent_id: str | None = None,
|
|
278
|
+
run_id: str | None = None, ts: datetime | str | None = None,
|
|
279
|
+
source: str = "", chunk_id: str | None = None) -> str:
|
|
280
|
+
cid = chunk_id or new_id()
|
|
281
|
+
ts_s = iso(parse_ts(ts) or datetime.now(timezone.utc)) if ts else iso(datetime.now(timezone.utc))
|
|
282
|
+
self.conn.execute(
|
|
283
|
+
"INSERT OR REPLACE INTO chunks(id, text, user_id, agent_id, run_id, ts, source, hash, tokens) "
|
|
284
|
+
"VALUES(?,?,?,?,?,?,?,?,?)",
|
|
285
|
+
(cid, text, user_id, agent_id, run_id, ts_s, source,
|
|
286
|
+
self.hasher.hash_text(text), token_estimate(text)))
|
|
287
|
+
self._maybe_commit()
|
|
288
|
+
return cid
|
|
289
|
+
|
|
290
|
+
def get_chunk(self, chunk_id: str) -> dict | None:
|
|
291
|
+
row = self.conn.execute("SELECT * FROM chunks WHERE id=?", (chunk_id,)).fetchone()
|
|
292
|
+
return dict(row) if row else None
|
|
293
|
+
|
|
294
|
+
def all_chunks(self, user_id: str | None = None) -> list[dict]:
|
|
295
|
+
if user_id:
|
|
296
|
+
rows = self.conn.execute("SELECT * FROM chunks WHERE user_id=? ORDER BY ts", (user_id,)).fetchall()
|
|
297
|
+
else:
|
|
298
|
+
rows = self.conn.execute("SELECT * FROM chunks ORDER BY ts").fetchall()
|
|
299
|
+
return [dict(r) for r in rows]
|
|
300
|
+
|
|
301
|
+
def quarantined_chunk_texts(self, user_id: str | None = None) -> list[str]:
|
|
302
|
+
"""Source texts of every quarantined fact (the tainted corpus used
|
|
303
|
+
by the MINJA contagion guard on the write path)."""
|
|
304
|
+
sql = ("SELECT DISTINCT c.text FROM chunks c "
|
|
305
|
+
"JOIN facts f ON f.source_id = c.id WHERE f.quarantined = 1")
|
|
306
|
+
args: tuple = ()
|
|
307
|
+
if user_id is not None:
|
|
308
|
+
sql += " AND c.user_id = ?"
|
|
309
|
+
args = (user_id,)
|
|
310
|
+
return [r[0] for r in self.conn.execute(sql, args).fetchall()]
|
|
311
|
+
|
|
312
|
+
# ------------------------------------------------------------- commits
|
|
313
|
+
def create_commit(self, message: str = "", branch: str | None = None,
|
|
314
|
+
parents: list[str] | None = None, n_facts: int = 0) -> str:
|
|
315
|
+
branch = branch or self.current_branch()
|
|
316
|
+
head = self.head(branch)
|
|
317
|
+
parents = parents if parents is not None else ([head] if head else [])
|
|
318
|
+
cid = new_id()
|
|
319
|
+
now = iso(datetime.now(timezone.utc))
|
|
320
|
+
parent_chains = [self.conn.execute(
|
|
321
|
+
"SELECT chain_hash FROM commits WHERE id=?", (p,)).fetchone() for p in parents]
|
|
322
|
+
chain = self.hasher.hash_json({
|
|
323
|
+
"commit": cid, "parents": parents, "message": message,
|
|
324
|
+
"ts": now,
|
|
325
|
+
"parent_chains": [r["chain_hash"] if r else "" for r in parent_chains],
|
|
326
|
+
})
|
|
327
|
+
self.conn.execute(
|
|
328
|
+
"INSERT INTO commits(id, parents, branch, message, ts, chain_hash, n_facts) VALUES(?,?,?,?,?,?,?)",
|
|
329
|
+
(cid, json.dumps(parents), branch, message, now, chain, n_facts))
|
|
330
|
+
if not parents:
|
|
331
|
+
self.conn.execute(
|
|
332
|
+
"INSERT OR REPLACE INTO branches(name, head, created) VALUES(?,?,?)",
|
|
333
|
+
(branch, cid, now))
|
|
334
|
+
else:
|
|
335
|
+
self.conn.execute("UPDATE branches SET head=? WHERE name=?", (cid, branch))
|
|
336
|
+
self._invalidate(branch)
|
|
337
|
+
self._maybe_commit()
|
|
338
|
+
return cid
|
|
339
|
+
|
|
340
|
+
def head(self, branch: str | None = None) -> str | None:
|
|
341
|
+
branch = branch or self.current_branch()
|
|
342
|
+
row = self.conn.execute("SELECT head FROM branches WHERE name=?", (branch,)).fetchone()
|
|
343
|
+
return row["head"] if row else None
|
|
344
|
+
|
|
345
|
+
def current_branch(self) -> str:
|
|
346
|
+
return self.kv_get("HEAD_BRANCH", "main") or "main"
|
|
347
|
+
|
|
348
|
+
def checkout(self, branch: str) -> None:
|
|
349
|
+
if not self.conn.execute("SELECT 1 FROM branches WHERE name=?", (branch,)).fetchone():
|
|
350
|
+
raise BranchError(f"unknown branch {branch!r}")
|
|
351
|
+
self.kv_set("HEAD_BRANCH", branch)
|
|
352
|
+
|
|
353
|
+
def create_branch(self, name: str, from_commit: str | None = None,
|
|
354
|
+
switch: bool = True) -> str:
|
|
355
|
+
if self.conn.execute("SELECT 1 FROM branches WHERE name=?", (name,)).fetchone():
|
|
356
|
+
raise BranchError(f"branch {name!r} already exists")
|
|
357
|
+
base = from_commit or self.head() or self.head(self.current_branch())
|
|
358
|
+
if base is None:
|
|
359
|
+
raise BranchError("cannot branch from empty history")
|
|
360
|
+
self.conn.execute(
|
|
361
|
+
"INSERT INTO branches(name, head, created) VALUES(?,?,?)",
|
|
362
|
+
(name, base, iso(datetime.now(timezone.utc))))
|
|
363
|
+
if switch:
|
|
364
|
+
self.kv_set("HEAD_BRANCH", name)
|
|
365
|
+
self._maybe_commit()
|
|
366
|
+
return base
|
|
367
|
+
|
|
368
|
+
def branches(self) -> list[dict]:
|
|
369
|
+
rows = self.conn.execute(
|
|
370
|
+
"SELECT b.name, b.head, b.created, c.ts AS head_ts, c.message "
|
|
371
|
+
"FROM branches b LEFT JOIN commits c ON c.id=b.head ORDER BY b.created").fetchall()
|
|
372
|
+
return [dict(r) for r in rows]
|
|
373
|
+
|
|
374
|
+
def log(self, branch: str | None = None, limit: int = 50) -> list[dict]:
|
|
375
|
+
cid = self.head(branch or self.current_branch())
|
|
376
|
+
out: list[dict] = []
|
|
377
|
+
seen = set()
|
|
378
|
+
queue = [cid] if cid else []
|
|
379
|
+
while queue and len(out) < limit:
|
|
380
|
+
cur = queue.pop(0)
|
|
381
|
+
if not cur or cur in seen:
|
|
382
|
+
continue
|
|
383
|
+
seen.add(cur)
|
|
384
|
+
row = self.conn.execute("SELECT * FROM commits WHERE id=?", (cur,)).fetchone()
|
|
385
|
+
if not row:
|
|
386
|
+
continue
|
|
387
|
+
out.append(dict(row))
|
|
388
|
+
queue.extend(json.loads(row["parents"]))
|
|
389
|
+
return out
|
|
390
|
+
|
|
391
|
+
def ancestry(self, commit_id: str) -> frozenset[str]:
|
|
392
|
+
cached = self._ancestry_cache.get(commit_id)
|
|
393
|
+
if cached is not None:
|
|
394
|
+
return cached
|
|
395
|
+
seen: set[str] = set()
|
|
396
|
+
queue = [commit_id]
|
|
397
|
+
while queue:
|
|
398
|
+
cur = queue.pop()
|
|
399
|
+
if cur in seen:
|
|
400
|
+
continue
|
|
401
|
+
seen.add(cur)
|
|
402
|
+
row = self.conn.execute("SELECT parents FROM commits WHERE id=?", (cur,)).fetchone()
|
|
403
|
+
if row:
|
|
404
|
+
queue.extend(json.loads(row["parents"]))
|
|
405
|
+
result = frozenset(seen)
|
|
406
|
+
if len(self._ancestry_cache) < 64:
|
|
407
|
+
self._ancestry_cache[commit_id] = result
|
|
408
|
+
return result
|
|
409
|
+
|
|
410
|
+
def commit(self, commit_id: str) -> dict | None:
|
|
411
|
+
row = self.conn.execute("SELECT * FROM commits WHERE id=?", (commit_id,)).fetchone()
|
|
412
|
+
return dict(row) if row else None
|
|
413
|
+
|
|
414
|
+
# ---------------------------------------------------------------- facts
|
|
415
|
+
def insert_fact(self, fact: Fact, commit_id: str | None = None) -> Fact:
|
|
416
|
+
if commit_id:
|
|
417
|
+
fact.birth_commit = commit_id
|
|
418
|
+
row = fact.to_row()
|
|
419
|
+
cols = ", ".join(row.keys())
|
|
420
|
+
ph = ", ".join("?" for _ in row)
|
|
421
|
+
self.conn.execute(f"INSERT INTO facts({cols}) VALUES({ph})", tuple(row.values()))
|
|
422
|
+
return fact
|
|
423
|
+
|
|
424
|
+
def insert_facts_bulk(self, facts: list[Fact], commit_id: str | None = None) -> int:
|
|
425
|
+
for f in facts:
|
|
426
|
+
if commit_id:
|
|
427
|
+
f.birth_commit = commit_id
|
|
428
|
+
rows = [f.to_row() for f in facts]
|
|
429
|
+
if not rows:
|
|
430
|
+
return 0
|
|
431
|
+
cols = ", ".join(rows[0].keys())
|
|
432
|
+
ph = ", ".join("?" for _ in rows[0])
|
|
433
|
+
self.conn.executemany(f"INSERT INTO facts({cols}) VALUES({ph})",
|
|
434
|
+
[tuple(r.values()) for r in rows])
|
|
435
|
+
return len(rows)
|
|
436
|
+
|
|
437
|
+
def update_commit_n_facts(self, commit_id: str, n_facts: int) -> None:
|
|
438
|
+
"""Update the n_facts counter on a commit (after cognition engine
|
|
439
|
+
appends derived facts post-creation)."""
|
|
440
|
+
if not commit_id:
|
|
441
|
+
return
|
|
442
|
+
self.conn.execute(
|
|
443
|
+
"UPDATE commits SET n_facts = n_facts + ? WHERE id=?",
|
|
444
|
+
(n_facts, commit_id))
|
|
445
|
+
self._maybe_commit()
|
|
446
|
+
|
|
447
|
+
def get_fact(self, fact_id: str) -> Fact | None:
|
|
448
|
+
row = self.conn.execute(f"SELECT {FACT_COLUMNS} FROM facts WHERE id=?", (fact_id,)).fetchone()
|
|
449
|
+
return Fact.from_row(dict(row)) if row else None
|
|
450
|
+
|
|
451
|
+
def get_facts(self, ids: list[str]) -> list[Fact]:
|
|
452
|
+
if not ids:
|
|
453
|
+
return []
|
|
454
|
+
out = []
|
|
455
|
+
for i in range(0, len(ids), 500):
|
|
456
|
+
batch = ids[i:i + 500]
|
|
457
|
+
q = f"SELECT {FACT_COLUMNS} FROM facts WHERE id IN ({','.join('?' * len(batch))})"
|
|
458
|
+
out.extend(Fact.from_row(dict(r)) for r in self.conn.execute(q, batch))
|
|
459
|
+
return out
|
|
460
|
+
|
|
461
|
+
def update_fact(self, fact_id: str, **fields) -> None:
|
|
462
|
+
if not fields:
|
|
463
|
+
return
|
|
464
|
+
if "provenance" in fields and isinstance(fields["provenance"], dict):
|
|
465
|
+
fields["provenance"] = json.dumps(fields["provenance"], default=str)
|
|
466
|
+
sets = ", ".join(f"{k}=?" for k in fields)
|
|
467
|
+
self.conn.execute(f"UPDATE facts SET {sets} WHERE id=?", (*fields.values(), fact_id))
|
|
468
|
+
self._maybe_commit()
|
|
469
|
+
|
|
470
|
+
def bump_access(self, fact_ids: list[str]) -> None:
|
|
471
|
+
for fid in fact_ids:
|
|
472
|
+
self.conn.execute("UPDATE facts SET access_count=access_count+1 WHERE id=?", (fid,))
|
|
473
|
+
self._maybe_commit()
|
|
474
|
+
|
|
475
|
+
def _fact_filters(self, user_id=None, agent_id=None, run_id=None, branch=None,
|
|
476
|
+
active=True, include_quarantined=False, subject=None,
|
|
477
|
+
relation=None, value=None, derived=None):
|
|
478
|
+
clauses, params = [], []
|
|
479
|
+
if active:
|
|
480
|
+
clauses.append("is_active=1")
|
|
481
|
+
if not include_quarantined:
|
|
482
|
+
clauses.append("quarantined=0")
|
|
483
|
+
if user_id is not None:
|
|
484
|
+
clauses.append("user_id=?"); params.append(user_id)
|
|
485
|
+
if agent_id is not None:
|
|
486
|
+
clauses.append("agent_id=?"); params.append(agent_id)
|
|
487
|
+
if run_id is not None:
|
|
488
|
+
clauses.append("run_id=?"); params.append(run_id)
|
|
489
|
+
if subject is not None:
|
|
490
|
+
clauses.append("subject=?"); params.append(subject)
|
|
491
|
+
if relation is not None:
|
|
492
|
+
clauses.append("relation=?"); params.append(relation)
|
|
493
|
+
if value is not None:
|
|
494
|
+
clauses.append("value=?"); params.append(value)
|
|
495
|
+
if derived is not None:
|
|
496
|
+
clauses.append("is_derived=?"); params.append(int(derived))
|
|
497
|
+
if branch is not None:
|
|
498
|
+
ids = self.active_ids(branch)
|
|
499
|
+
if not ids:
|
|
500
|
+
clauses.append("1=0")
|
|
501
|
+
else:
|
|
502
|
+
# membership filtering applied post-hoc for large sets
|
|
503
|
+
pass
|
|
504
|
+
return clauses, params, (branch if branch is not None else None)
|
|
505
|
+
|
|
506
|
+
def query_facts(self, *, subject=None, relation=None, value=None, user_id=None,
|
|
507
|
+
agent_id=None, run_id=None, branch=None, active=True,
|
|
508
|
+
include_quarantined=False, derived=None, order="valid_from",
|
|
509
|
+
limit=None) -> list[Fact]:
|
|
510
|
+
clauses, params, branch_filter = self._fact_filters(
|
|
511
|
+
user_id, agent_id, run_id, branch, active, include_quarantined,
|
|
512
|
+
subject, relation, value, derived)
|
|
513
|
+
where = (" WHERE " + " AND ".join(clauses)) if clauses else ""
|
|
514
|
+
q = f"SELECT {FACT_COLUMNS} FROM facts{where} ORDER BY {order}"
|
|
515
|
+
if limit:
|
|
516
|
+
q += f" LIMIT {int(limit)}"
|
|
517
|
+
rows = self.conn.execute(q, params).fetchall()
|
|
518
|
+
facts = [Fact.from_row(dict(r)) for r in rows]
|
|
519
|
+
if branch_filter is not None:
|
|
520
|
+
ids = self.active_ids(branch_filter)
|
|
521
|
+
facts = [f for f in facts if f.id in ids]
|
|
522
|
+
return facts
|
|
523
|
+
|
|
524
|
+
def active_facts(self, **kw) -> list[Fact]:
|
|
525
|
+
return self.query_facts(**kw)
|
|
526
|
+
|
|
527
|
+
def history_of(self, subject: str, relation: str, user_id: str | None = None,
|
|
528
|
+
include_inactive=True) -> list[Fact]:
|
|
529
|
+
clauses, params, _ = self._fact_filters(
|
|
530
|
+
user_id=user_id, active=not include_inactive)
|
|
531
|
+
extra = "subject=? AND relation=?" + ((" AND " + " AND ".join(clauses)) if clauses else "")
|
|
532
|
+
q = f"SELECT {FACT_COLUMNS} FROM facts WHERE {extra} ORDER BY valid_from, tx_from"
|
|
533
|
+
rows = self.conn.execute(q, (subject, relation, *params)).fetchall()
|
|
534
|
+
return [Fact.from_row(dict(r)) for r in rows]
|
|
535
|
+
|
|
536
|
+
def facts_about(self, entity: str, user_id: str | None = None,
|
|
537
|
+
active: bool = True) -> list[Fact]:
|
|
538
|
+
"""Facts where entity is subject OR value (1-hop associative recall)."""
|
|
539
|
+
clauses, params, _ = self._fact_filters(user_id=user_id, active=active)
|
|
540
|
+
extra = "(subject=? OR value=?)" + ((" AND " + " AND ".join(clauses)) if clauses else "")
|
|
541
|
+
q = f"SELECT {FACT_COLUMNS} FROM facts WHERE {extra}"
|
|
542
|
+
rows = self.conn.execute(q, (entity, entity, *params)).fetchall()
|
|
543
|
+
return [Fact.from_row(dict(r)) for r in rows]
|
|
544
|
+
|
|
545
|
+
def temporal_window(self, start: str | None, end: str | None,
|
|
546
|
+
user_id: str | None = None, field: str = "valid",
|
|
547
|
+
active: bool = True) -> list[Fact]:
|
|
548
|
+
"""Zep-compatible temporal queries. field: 'valid' (reality) or 'tx' (recorded).
|
|
549
|
+
|
|
550
|
+
Valid-time uses interval-overlap semantics: a fact matches if its
|
|
551
|
+
[valid_from, valid_to] window intersects [start, end] — so asking
|
|
552
|
+
"where did Alice work in 2025?" retrieves an employment that began
|
|
553
|
+
in 2024 and ended in 2026. Transaction-time uses point semantics.
|
|
554
|
+
"""
|
|
555
|
+
clauses, fparams, _ = self._fact_filters(user_id=user_id, active=active)
|
|
556
|
+
cond: list[str] = []
|
|
557
|
+
cparams: list = []
|
|
558
|
+
if field == "valid":
|
|
559
|
+
if start:
|
|
560
|
+
cond.append("(valid_to IS NULL OR valid_to>=?)")
|
|
561
|
+
cparams.append(start[:10])
|
|
562
|
+
if end:
|
|
563
|
+
cond.append("valid_from<=?")
|
|
564
|
+
cparams.append(end[:10])
|
|
565
|
+
order = "valid_from"
|
|
566
|
+
else:
|
|
567
|
+
if start:
|
|
568
|
+
cond.append("tx_from>=?"); cparams.append(start[:10])
|
|
569
|
+
if end:
|
|
570
|
+
cond.append("tx_from<=?"); cparams.append(end[:10])
|
|
571
|
+
order = "tx_from"
|
|
572
|
+
params = cparams + fparams
|
|
573
|
+
where = " AND ".join(cond + clauses) if (cond + clauses) else ""
|
|
574
|
+
q = f"SELECT {FACT_COLUMNS} FROM facts{' WHERE ' + where if where else ''} ORDER BY {order}"
|
|
575
|
+
rows = self.conn.execute(q, params).fetchall()
|
|
576
|
+
return [Fact.from_row(dict(r)) for r in rows]
|
|
577
|
+
|
|
578
|
+
def count_facts(self, user_id: str | None = None, active_only: bool = True) -> int:
|
|
579
|
+
clauses, params, _ = self._fact_filters(user_id=user_id, active=active_only)
|
|
580
|
+
where = (" WHERE " + " AND ".join(clauses)) if clauses else ""
|
|
581
|
+
row = self.conn.execute(f"SELECT COUNT(*) c FROM facts{where}", params).fetchone()
|
|
582
|
+
return int(row["c"])
|
|
583
|
+
|
|
584
|
+
# ---------------------------------------------------------- active set
|
|
585
|
+
def active_ids(self, branch: str) -> frozenset[str]:
|
|
586
|
+
cached = self._active_cache.get(branch)
|
|
587
|
+
if cached is not None:
|
|
588
|
+
return cached
|
|
589
|
+
head = self.head(branch)
|
|
590
|
+
if head is None:
|
|
591
|
+
return frozenset()
|
|
592
|
+
anc = self.ancestry(head)
|
|
593
|
+
rows = self.conn.execute(
|
|
594
|
+
"SELECT id, birth_commit, retired_commit FROM facts "
|
|
595
|
+
"WHERE birth_commit IS NOT NULL").fetchall()
|
|
596
|
+
ids = frozenset(
|
|
597
|
+
r["id"] for r in rows
|
|
598
|
+
if r["birth_commit"] in anc and (
|
|
599
|
+
not r["retired_commit"] or r["retired_commit"] not in anc))
|
|
600
|
+
if len(self._active_cache) < 16:
|
|
601
|
+
self._active_cache[branch] = ids
|
|
602
|
+
return ids
|
|
603
|
+
|
|
604
|
+
def _invalidate(self, branch: str | None = None) -> None:
|
|
605
|
+
if branch:
|
|
606
|
+
self._active_cache.pop(branch, None)
|
|
607
|
+
else:
|
|
608
|
+
self._active_cache.clear()
|
|
609
|
+
self._ancestry_cache.clear()
|
|
610
|
+
|
|
611
|
+
# ---------------------------------------------------------------- edges
|
|
612
|
+
def add_edge(self, src: str, dst: str, kind: str, meta: dict | None = None) -> None:
|
|
613
|
+
self.conn.execute(
|
|
614
|
+
"INSERT OR REPLACE INTO edges(src, dst, kind, meta, created) VALUES(?,?,?,?,?)",
|
|
615
|
+
(src, dst, kind, json.dumps(meta or {}), iso(datetime.now(timezone.utc))))
|
|
616
|
+
self._maybe_commit()
|
|
617
|
+
|
|
618
|
+
def edges_of(self, fact_id: str, kind: str | None = None, direction: str = "out") -> list[dict]:
|
|
619
|
+
outs, ins = [], []
|
|
620
|
+
if direction in ("out", "both"):
|
|
621
|
+
q = "SELECT * FROM edges WHERE src=?" + (" AND kind=?" if kind else "")
|
|
622
|
+
rows = self.conn.execute(q, (fact_id, kind) if kind else (fact_id,)).fetchall()
|
|
623
|
+
outs = [dict(r, dir="out") for r in rows]
|
|
624
|
+
if direction in ("in", "both"):
|
|
625
|
+
q = "SELECT * FROM edges WHERE dst=?" + (" AND kind=?" if kind else "")
|
|
626
|
+
rows = self.conn.execute(q, (fact_id, kind) if kind else (fact_id,)).fetchall()
|
|
627
|
+
ins = [dict(r, dir="in") for r in rows]
|
|
628
|
+
return outs + ins
|
|
629
|
+
|
|
630
|
+
def edges_of_many(self, fact_ids: list[str], kind: str | None = None) -> list[dict]:
|
|
631
|
+
"""Edges among a set of fact ids (both directions), batched."""
|
|
632
|
+
if not fact_ids:
|
|
633
|
+
return []
|
|
634
|
+
out: list[dict] = []
|
|
635
|
+
B = 200
|
|
636
|
+
for i in range(0, len(fact_ids), B):
|
|
637
|
+
chunk = fact_ids[i:i + B]
|
|
638
|
+
qm = ",".join("?" * len(chunk))
|
|
639
|
+
q = (f"SELECT * FROM edges WHERE src IN ({qm}) AND dst IN ({qm})"
|
|
640
|
+
+ (f" AND kind=?" if kind else ""))
|
|
641
|
+
rows = self.conn.execute(q, chunk + chunk + ([kind] if kind else []))
|
|
642
|
+
out.extend(dict(r) for r in rows)
|
|
643
|
+
return out
|
|
644
|
+
|
|
645
|
+
# ------------------------------------------------------------ integrity
|
|
646
|
+
def active_fact_hashes(self, user_id: str | None = None) -> list[tuple[str, str]]:
|
|
647
|
+
clauses, params, _ = self._fact_filters(user_id=user_id, active=True)
|
|
648
|
+
where = (" WHERE " + " AND ".join(clauses)) if clauses else ""
|
|
649
|
+
rows = self.conn.execute(
|
|
650
|
+
f"SELECT id, source_hash FROM facts{where} ORDER BY id", params).fetchall()
|
|
651
|
+
return [(r["id"], r["source_hash"] or "") for r in rows]
|
|
652
|
+
|
|
653
|
+
def stats(self) -> dict:
|
|
654
|
+
def one(q, *p):
|
|
655
|
+
return int(self.conn.execute(q, p).fetchone()[0])
|
|
656
|
+
return {
|
|
657
|
+
"facts": one("SELECT COUNT(*) FROM facts"),
|
|
658
|
+
"active_facts": one("SELECT COUNT(*) FROM facts WHERE is_active=1 AND quarantined=0"),
|
|
659
|
+
"quarantined": one("SELECT COUNT(*) FROM facts WHERE quarantined=1"),
|
|
660
|
+
"derived": one("SELECT COUNT(*) FROM facts WHERE is_derived=1"),
|
|
661
|
+
"chunks": one("SELECT COUNT(*) FROM chunks"),
|
|
662
|
+
"edges": one("SELECT COUNT(*) FROM edges"),
|
|
663
|
+
"commits": one("SELECT COUNT(*) FROM commits"),
|
|
664
|
+
"branches": one("SELECT COUNT(*) FROM branches"),
|
|
665
|
+
"long_term": one("SELECT COUNT(*) FROM facts WHERE memory_type='long_term'"),
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
def close(self) -> None:
|
|
669
|
+
try:
|
|
670
|
+
self.conn.commit()
|
|
671
|
+
self.checkpoint("TRUNCATE")
|
|
672
|
+
self.conn.close()
|
|
673
|
+
except Exception:
|
|
674
|
+
pass
|
|
675
|
+
|
|
676
|
+
def __enter__(self) -> "TraceStore":
|
|
677
|
+
return self
|
|
678
|
+
|
|
679
|
+
def __exit__(self, *exc) -> None:
|
|
680
|
+
self.close()
|