cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/trace/store.py ADDED
@@ -0,0 +1,680 @@
1
+ """The Symbolic Trace — bi-temporal fact store on SQLite (WAL).
2
+
3
+ Implements the Section 1.1 data model: Subject-Relation-Value triples
4
+ with valid/transaction time, contradiction + temporal + provenance
5
+ edges, source chunks, Memory-Git commits (hash-chained) and branches.
6
+ SQLite is the embedded substrate for local/edge operation; the store is
7
+ designed so an ArcadeDB backend can replace it behind the same API
8
+ (the plan's production graph engine).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import sqlite3
15
+ import threading
16
+ from datetime import datetime, timezone
17
+
18
+ from cortexm.errors import BranchError, StoreError
19
+ from cortexm.security.hashes import HashProvider
20
+ from cortexm.trace.fact import Fact
21
+ from cortexm.util import iso, new_id, parse_ts, token_estimate
22
+
23
+ SCHEMA = """
24
+ CREATE TABLE IF NOT EXISTS facts (
25
+ id TEXT PRIMARY KEY,
26
+ subject TEXT NOT NULL, relation TEXT NOT NULL, value TEXT NOT NULL,
27
+ valid_from TEXT NOT NULL, valid_to TEXT,
28
+ tx_from TEXT NOT NULL, tx_to TEXT,
29
+ confidence REAL DEFAULT 0.8,
30
+ source_hash TEXT DEFAULT '', source_id TEXT DEFAULT '',
31
+ user_id TEXT DEFAULT 'default', agent_id TEXT, run_id TEXT,
32
+ memory_type TEXT DEFAULT 'short_term',
33
+ access_count INTEGER DEFAULT 0, reinforcement INTEGER DEFAULT 1,
34
+ is_active INTEGER DEFAULT 1, is_derived INTEGER DEFAULT 0,
35
+ quarantined INTEGER DEFAULT 0,
36
+ birth_commit TEXT, retired_commit TEXT,
37
+ provenance TEXT DEFAULT '{}'
38
+ );
39
+ CREATE INDEX IF NOT EXISTS idx_facts_sr ON facts(subject, relation);
40
+ CREATE INDEX IF NOT EXISTS idx_facts_user ON facts(user_id);
41
+ CREATE INDEX IF NOT EXISTS idx_facts_active ON facts(is_active);
42
+ CREATE INDEX IF NOT EXISTS idx_facts_rel ON facts(relation);
43
+ CREATE INDEX IF NOT EXISTS idx_facts_valid ON facts(valid_from);
44
+ CREATE INDEX IF NOT EXISTS idx_facts_birth ON facts(birth_commit);
45
+ -- composite indexes added for the v2 SPARQL + REST query paths
46
+ CREATE INDEX IF NOT EXISTS idx_facts_user_active ON facts(user_id, is_active);
47
+ CREATE INDEX IF NOT EXISTS idx_facts_value ON facts(value);
48
+ CREATE INDEX IF NOT EXISTS idx_facts_subject_value ON facts(subject, value);
49
+
50
+ CREATE TABLE IF NOT EXISTS edges (
51
+ src TEXT NOT NULL, dst TEXT NOT NULL, kind TEXT NOT NULL,
52
+ meta TEXT DEFAULT '{}', created TEXT,
53
+ PRIMARY KEY (src, dst, kind)
54
+ );
55
+ CREATE INDEX IF NOT EXISTS idx_edges_src ON edges(src);
56
+ CREATE INDEX IF NOT EXISTS idx_edges_dst ON edges(dst);
57
+ -- kind index — critical for SPARQL `?a edge:CAUSAL ?b` queries that
58
+ -- post-filter on the kind column
59
+ CREATE INDEX IF NOT EXISTS idx_edges_kind ON edges(kind);
60
+
61
+ CREATE TABLE IF NOT EXISTS chunks (
62
+ id TEXT PRIMARY KEY, text TEXT NOT NULL,
63
+ user_id TEXT, agent_id TEXT, run_id TEXT,
64
+ ts TEXT, source TEXT DEFAULT '', hash TEXT DEFAULT '', tokens INTEGER DEFAULT 0
65
+ );
66
+ CREATE INDEX IF NOT EXISTS idx_chunks_user ON chunks(user_id);
67
+
68
+ CREATE TABLE IF NOT EXISTS commits (
69
+ id TEXT PRIMARY KEY, parents TEXT DEFAULT '[]', branch TEXT,
70
+ message TEXT DEFAULT '', ts TEXT, chain_hash TEXT, n_facts INTEGER DEFAULT 0
71
+ );
72
+ CREATE TABLE IF NOT EXISTS branches (
73
+ name TEXT PRIMARY KEY, head TEXT NOT NULL, created TEXT
74
+ );
75
+ CREATE TABLE IF NOT EXISTS kv (k TEXT PRIMARY KEY, v TEXT);
76
+ """
77
+
78
+ FACT_COLUMNS = ("id, subject, relation, value, valid_from, valid_to, tx_from, tx_to, "
79
+ "confidence, source_hash, source_id, user_id, agent_id, run_id, "
80
+ "memory_type, access_count, reinforcement, is_active, is_derived, "
81
+ "quarantined, birth_commit, retired_commit, provenance")
82
+
83
+
84
+ class _SafeCursor:
85
+ """Cursor-like object over eagerly-materialized rows (thread-safe)."""
86
+
87
+ def __init__(self, rows, rowcount, lastrowid, description) -> None:
88
+ self._rows = rows
89
+ self._iter = iter(rows)
90
+ self.rowcount = rowcount
91
+ self.lastrowid = lastrowid
92
+ self.description = description
93
+ self.arraysize = 1
94
+
95
+ def fetchone(self):
96
+ try:
97
+ return next(self._iter)
98
+ except StopIteration:
99
+ return None
100
+
101
+ def fetchall(self):
102
+ rest = list(self._iter)
103
+ self._iter = iter([])
104
+ return rest
105
+
106
+ def fetchmany(self, size=None):
107
+ out = []
108
+ for _ in range(size or 1):
109
+ try:
110
+ out.append(next(self._iter))
111
+ except StopIteration:
112
+ break
113
+ return out
114
+
115
+ def __iter__(self):
116
+ return self
117
+
118
+ def __next__(self):
119
+ return next(self._iter)
120
+
121
+ def close(self) -> None:
122
+ pass
123
+
124
+
125
+ class SafeConnection:
126
+ """Serializes every statement on one SQLite connection.
127
+
128
+ SQLite connections are not safe for concurrent cursor use even with
129
+ ``check_same_thread=False`` — interleaved commit/iterate produces
130
+ InterfaceError('bad parameter or other API misuse'). This wrapper
131
+ holds an RLock across execute+materialize and across commit, making
132
+ the whole TraceStore safe under multi-threaded load (REST server,
133
+ concurrent writers). Eager materialization keeps semantics: callers
134
+ only use fetchone/fetchall/iteration/rowcount/lastrowid.
135
+ """
136
+
137
+ def __init__(self, conn) -> None:
138
+ self._conn = conn
139
+ self._lock = threading.RLock()
140
+
141
+ def execute(self, sql, params=()):
142
+ with self._lock:
143
+ cur = self._conn.execute(sql, params)
144
+ try:
145
+ rows = cur.fetchall()
146
+ except Exception:
147
+ rows = []
148
+ rc, lrid, desc = cur.rowcount, cur.lastrowid, cur.description
149
+ cur.close()
150
+ return _SafeCursor(rows, rc, lrid, desc)
151
+
152
+ def executemany(self, sql, seq):
153
+ with self._lock:
154
+ cur = self._conn.executemany(sql, seq)
155
+ rc = cur.rowcount
156
+ cur.close()
157
+ return _SafeCursor([], rc, None, None)
158
+
159
+ def executescript(self, script):
160
+ with self._lock:
161
+ return self._conn.executescript(script)
162
+
163
+ def commit(self):
164
+ with self._lock:
165
+ self._conn.commit()
166
+
167
+ def rollback(self):
168
+ with self._lock:
169
+ self._conn.rollback()
170
+
171
+ def close(self):
172
+ with self._lock:
173
+ self._conn.close()
174
+
175
+ @property
176
+ def row_factory(self):
177
+ return self._conn.row_factory
178
+
179
+ @row_factory.setter
180
+ def row_factory(self, v):
181
+ with self._lock:
182
+ self._conn.row_factory = v
183
+
184
+ @property
185
+ def in_transaction(self):
186
+ return self._conn.in_transaction
187
+
188
+
189
+ class TraceStore:
190
+ def __init__(self, db_path: str = ":memory:", provider: HashProvider | None = None,
191
+ wal_sync: str = "normal") -> None:
192
+ self.db_path = db_path
193
+ self.hasher = provider or HashProvider()
194
+ mem = db_path in (":memory:", None, "")
195
+ self.conn = SafeConnection(
196
+ sqlite3.connect(db_path or ":memory:", check_same_thread=False))
197
+ if not mem:
198
+ # Aeon-inspired crash-recoverable write path:
199
+ # journal_mode=WAL — readers never block the writer, and a
200
+ # torn write rolls back cleanly on reopen.
201
+ # synchronous — NORMAL: commits survive process crash
202
+ # (SIGKILL) at full speed; FULL additionally
203
+ # survives OS/power loss at fsync cost.
204
+ self.conn.execute("PRAGMA journal_mode=WAL")
205
+ self.conn.execute(
206
+ "PRAGMA synchronous="
207
+ + ("FULL" if str(wal_sync).lower() == "full" else "NORMAL"))
208
+ self.conn.row_factory = sqlite3.Row
209
+ self.conn.executescript(SCHEMA)
210
+ self._ancestry_cache: dict[str, frozenset[str]] = {}
211
+ self._active_cache: dict[str, frozenset[str]] = {}
212
+ self._batching = False
213
+ self._ensure_genesis()
214
+
215
+ def checkpoint(self, mode: str = "TRUNCATE") -> None:
216
+ """Fold the WAL back into the main db file (shrink + fast reopen)."""
217
+ if self.db_path in (":memory:", None, ""):
218
+ return
219
+ try:
220
+ self.conn.execute(f"PRAGMA wal_checkpoint({mode})")
221
+ except Exception:
222
+ pass
223
+
224
+ def begin_batch(self) -> None:
225
+ self._batching = True
226
+
227
+ def end_batch(self) -> None:
228
+ self._batching = False
229
+ self.conn.commit()
230
+
231
+ def _maybe_commit(self) -> None:
232
+ if not self._batching:
233
+ self.conn.commit()
234
+
235
+ @property
236
+ def batching(self) -> bool:
237
+ return self._batching
238
+
239
+ # ------------------------------------------------------------------ kv
240
+ def kv_get(self, key: str, default: str | None = None) -> str | None:
241
+ row = self.conn.execute("SELECT v FROM kv WHERE k=?", (key,)).fetchone()
242
+ return row["v"] if row else default
243
+
244
+ def kv_set(self, key: str, value: str) -> None:
245
+ self.conn.execute(
246
+ "INSERT INTO kv(k, v) VALUES(?, ?) ON CONFLICT(k) DO UPDATE SET v=excluded.v",
247
+ (key, value))
248
+ self.conn.commit()
249
+
250
+ def iter_kv(self, prefix: str = ""):
251
+ """Yield (key, value) pairs whose key starts with ``prefix``."""
252
+ cur = self.conn.execute(
253
+ "SELECT k, v FROM kv WHERE k LIKE ? ORDER BY k",
254
+ (prefix + "%",))
255
+ for row in cur:
256
+ yield row["k"], row["v"]
257
+
258
+ def kv_delete(self, key: str) -> None:
259
+ self.conn.execute("DELETE FROM kv WHERE k=?", (key,))
260
+ self.conn.commit()
261
+
262
+ # -------------------------------------------------------------- genesis
263
+ def _ensure_genesis(self) -> None:
264
+ if not self.conn.execute("SELECT 1 FROM branches LIMIT 1").fetchone():
265
+ cid = new_id()
266
+ now = iso(datetime.utcnow().__class__.now() if False else datetime.utcnow()) if False else iso(datetime.now()) # noqa
267
+ chain = self.hasher.hash_text("genesis:" + cid)
268
+ self.conn.execute(
269
+ "INSERT INTO commits(id, parents, branch, message, ts, chain_hash) VALUES(?,?,?,?,?,?)",
270
+ (cid, "[]", "main", "genesis", now, chain))
271
+ self.conn.execute(
272
+ "INSERT INTO branches(name, head, created) VALUES(?,?,?)", ("main", cid, now))
273
+ self.kv_set("HEAD_BRANCH", "main")
274
+ self.kv_set("SCHEMA_VERSION", "1")
275
+
276
+ # -------------------------------------------------------------- chunks
277
+ def add_chunk(self, text: str, *, user_id: str = "default", agent_id: str | None = None,
278
+ run_id: str | None = None, ts: datetime | str | None = None,
279
+ source: str = "", chunk_id: str | None = None) -> str:
280
+ cid = chunk_id or new_id()
281
+ ts_s = iso(parse_ts(ts) or datetime.now(timezone.utc)) if ts else iso(datetime.now(timezone.utc))
282
+ self.conn.execute(
283
+ "INSERT OR REPLACE INTO chunks(id, text, user_id, agent_id, run_id, ts, source, hash, tokens) "
284
+ "VALUES(?,?,?,?,?,?,?,?,?)",
285
+ (cid, text, user_id, agent_id, run_id, ts_s, source,
286
+ self.hasher.hash_text(text), token_estimate(text)))
287
+ self._maybe_commit()
288
+ return cid
289
+
290
+ def get_chunk(self, chunk_id: str) -> dict | None:
291
+ row = self.conn.execute("SELECT * FROM chunks WHERE id=?", (chunk_id,)).fetchone()
292
+ return dict(row) if row else None
293
+
294
+ def all_chunks(self, user_id: str | None = None) -> list[dict]:
295
+ if user_id:
296
+ rows = self.conn.execute("SELECT * FROM chunks WHERE user_id=? ORDER BY ts", (user_id,)).fetchall()
297
+ else:
298
+ rows = self.conn.execute("SELECT * FROM chunks ORDER BY ts").fetchall()
299
+ return [dict(r) for r in rows]
300
+
301
+ def quarantined_chunk_texts(self, user_id: str | None = None) -> list[str]:
302
+ """Source texts of every quarantined fact (the tainted corpus used
303
+ by the MINJA contagion guard on the write path)."""
304
+ sql = ("SELECT DISTINCT c.text FROM chunks c "
305
+ "JOIN facts f ON f.source_id = c.id WHERE f.quarantined = 1")
306
+ args: tuple = ()
307
+ if user_id is not None:
308
+ sql += " AND c.user_id = ?"
309
+ args = (user_id,)
310
+ return [r[0] for r in self.conn.execute(sql, args).fetchall()]
311
+
312
+ # ------------------------------------------------------------- commits
313
+ def create_commit(self, message: str = "", branch: str | None = None,
314
+ parents: list[str] | None = None, n_facts: int = 0) -> str:
315
+ branch = branch or self.current_branch()
316
+ head = self.head(branch)
317
+ parents = parents if parents is not None else ([head] if head else [])
318
+ cid = new_id()
319
+ now = iso(datetime.now(timezone.utc))
320
+ parent_chains = [self.conn.execute(
321
+ "SELECT chain_hash FROM commits WHERE id=?", (p,)).fetchone() for p in parents]
322
+ chain = self.hasher.hash_json({
323
+ "commit": cid, "parents": parents, "message": message,
324
+ "ts": now,
325
+ "parent_chains": [r["chain_hash"] if r else "" for r in parent_chains],
326
+ })
327
+ self.conn.execute(
328
+ "INSERT INTO commits(id, parents, branch, message, ts, chain_hash, n_facts) VALUES(?,?,?,?,?,?,?)",
329
+ (cid, json.dumps(parents), branch, message, now, chain, n_facts))
330
+ if not parents:
331
+ self.conn.execute(
332
+ "INSERT OR REPLACE INTO branches(name, head, created) VALUES(?,?,?)",
333
+ (branch, cid, now))
334
+ else:
335
+ self.conn.execute("UPDATE branches SET head=? WHERE name=?", (cid, branch))
336
+ self._invalidate(branch)
337
+ self._maybe_commit()
338
+ return cid
339
+
340
+ def head(self, branch: str | None = None) -> str | None:
341
+ branch = branch or self.current_branch()
342
+ row = self.conn.execute("SELECT head FROM branches WHERE name=?", (branch,)).fetchone()
343
+ return row["head"] if row else None
344
+
345
+ def current_branch(self) -> str:
346
+ return self.kv_get("HEAD_BRANCH", "main") or "main"
347
+
348
+ def checkout(self, branch: str) -> None:
349
+ if not self.conn.execute("SELECT 1 FROM branches WHERE name=?", (branch,)).fetchone():
350
+ raise BranchError(f"unknown branch {branch!r}")
351
+ self.kv_set("HEAD_BRANCH", branch)
352
+
353
+ def create_branch(self, name: str, from_commit: str | None = None,
354
+ switch: bool = True) -> str:
355
+ if self.conn.execute("SELECT 1 FROM branches WHERE name=?", (name,)).fetchone():
356
+ raise BranchError(f"branch {name!r} already exists")
357
+ base = from_commit or self.head() or self.head(self.current_branch())
358
+ if base is None:
359
+ raise BranchError("cannot branch from empty history")
360
+ self.conn.execute(
361
+ "INSERT INTO branches(name, head, created) VALUES(?,?,?)",
362
+ (name, base, iso(datetime.now(timezone.utc))))
363
+ if switch:
364
+ self.kv_set("HEAD_BRANCH", name)
365
+ self._maybe_commit()
366
+ return base
367
+
368
+ def branches(self) -> list[dict]:
369
+ rows = self.conn.execute(
370
+ "SELECT b.name, b.head, b.created, c.ts AS head_ts, c.message "
371
+ "FROM branches b LEFT JOIN commits c ON c.id=b.head ORDER BY b.created").fetchall()
372
+ return [dict(r) for r in rows]
373
+
374
+ def log(self, branch: str | None = None, limit: int = 50) -> list[dict]:
375
+ cid = self.head(branch or self.current_branch())
376
+ out: list[dict] = []
377
+ seen = set()
378
+ queue = [cid] if cid else []
379
+ while queue and len(out) < limit:
380
+ cur = queue.pop(0)
381
+ if not cur or cur in seen:
382
+ continue
383
+ seen.add(cur)
384
+ row = self.conn.execute("SELECT * FROM commits WHERE id=?", (cur,)).fetchone()
385
+ if not row:
386
+ continue
387
+ out.append(dict(row))
388
+ queue.extend(json.loads(row["parents"]))
389
+ return out
390
+
391
+ def ancestry(self, commit_id: str) -> frozenset[str]:
392
+ cached = self._ancestry_cache.get(commit_id)
393
+ if cached is not None:
394
+ return cached
395
+ seen: set[str] = set()
396
+ queue = [commit_id]
397
+ while queue:
398
+ cur = queue.pop()
399
+ if cur in seen:
400
+ continue
401
+ seen.add(cur)
402
+ row = self.conn.execute("SELECT parents FROM commits WHERE id=?", (cur,)).fetchone()
403
+ if row:
404
+ queue.extend(json.loads(row["parents"]))
405
+ result = frozenset(seen)
406
+ if len(self._ancestry_cache) < 64:
407
+ self._ancestry_cache[commit_id] = result
408
+ return result
409
+
410
+ def commit(self, commit_id: str) -> dict | None:
411
+ row = self.conn.execute("SELECT * FROM commits WHERE id=?", (commit_id,)).fetchone()
412
+ return dict(row) if row else None
413
+
414
+ # ---------------------------------------------------------------- facts
415
+ def insert_fact(self, fact: Fact, commit_id: str | None = None) -> Fact:
416
+ if commit_id:
417
+ fact.birth_commit = commit_id
418
+ row = fact.to_row()
419
+ cols = ", ".join(row.keys())
420
+ ph = ", ".join("?" for _ in row)
421
+ self.conn.execute(f"INSERT INTO facts({cols}) VALUES({ph})", tuple(row.values()))
422
+ return fact
423
+
424
+ def insert_facts_bulk(self, facts: list[Fact], commit_id: str | None = None) -> int:
425
+ for f in facts:
426
+ if commit_id:
427
+ f.birth_commit = commit_id
428
+ rows = [f.to_row() for f in facts]
429
+ if not rows:
430
+ return 0
431
+ cols = ", ".join(rows[0].keys())
432
+ ph = ", ".join("?" for _ in rows[0])
433
+ self.conn.executemany(f"INSERT INTO facts({cols}) VALUES({ph})",
434
+ [tuple(r.values()) for r in rows])
435
+ return len(rows)
436
+
437
+ def update_commit_n_facts(self, commit_id: str, n_facts: int) -> None:
438
+ """Update the n_facts counter on a commit (after cognition engine
439
+ appends derived facts post-creation)."""
440
+ if not commit_id:
441
+ return
442
+ self.conn.execute(
443
+ "UPDATE commits SET n_facts = n_facts + ? WHERE id=?",
444
+ (n_facts, commit_id))
445
+ self._maybe_commit()
446
+
447
+ def get_fact(self, fact_id: str) -> Fact | None:
448
+ row = self.conn.execute(f"SELECT {FACT_COLUMNS} FROM facts WHERE id=?", (fact_id,)).fetchone()
449
+ return Fact.from_row(dict(row)) if row else None
450
+
451
+ def get_facts(self, ids: list[str]) -> list[Fact]:
452
+ if not ids:
453
+ return []
454
+ out = []
455
+ for i in range(0, len(ids), 500):
456
+ batch = ids[i:i + 500]
457
+ q = f"SELECT {FACT_COLUMNS} FROM facts WHERE id IN ({','.join('?' * len(batch))})"
458
+ out.extend(Fact.from_row(dict(r)) for r in self.conn.execute(q, batch))
459
+ return out
460
+
461
+ def update_fact(self, fact_id: str, **fields) -> None:
462
+ if not fields:
463
+ return
464
+ if "provenance" in fields and isinstance(fields["provenance"], dict):
465
+ fields["provenance"] = json.dumps(fields["provenance"], default=str)
466
+ sets = ", ".join(f"{k}=?" for k in fields)
467
+ self.conn.execute(f"UPDATE facts SET {sets} WHERE id=?", (*fields.values(), fact_id))
468
+ self._maybe_commit()
469
+
470
+ def bump_access(self, fact_ids: list[str]) -> None:
471
+ for fid in fact_ids:
472
+ self.conn.execute("UPDATE facts SET access_count=access_count+1 WHERE id=?", (fid,))
473
+ self._maybe_commit()
474
+
475
+ def _fact_filters(self, user_id=None, agent_id=None, run_id=None, branch=None,
476
+ active=True, include_quarantined=False, subject=None,
477
+ relation=None, value=None, derived=None):
478
+ clauses, params = [], []
479
+ if active:
480
+ clauses.append("is_active=1")
481
+ if not include_quarantined:
482
+ clauses.append("quarantined=0")
483
+ if user_id is not None:
484
+ clauses.append("user_id=?"); params.append(user_id)
485
+ if agent_id is not None:
486
+ clauses.append("agent_id=?"); params.append(agent_id)
487
+ if run_id is not None:
488
+ clauses.append("run_id=?"); params.append(run_id)
489
+ if subject is not None:
490
+ clauses.append("subject=?"); params.append(subject)
491
+ if relation is not None:
492
+ clauses.append("relation=?"); params.append(relation)
493
+ if value is not None:
494
+ clauses.append("value=?"); params.append(value)
495
+ if derived is not None:
496
+ clauses.append("is_derived=?"); params.append(int(derived))
497
+ if branch is not None:
498
+ ids = self.active_ids(branch)
499
+ if not ids:
500
+ clauses.append("1=0")
501
+ else:
502
+ # membership filtering applied post-hoc for large sets
503
+ pass
504
+ return clauses, params, (branch if branch is not None else None)
505
+
506
+ def query_facts(self, *, subject=None, relation=None, value=None, user_id=None,
507
+ agent_id=None, run_id=None, branch=None, active=True,
508
+ include_quarantined=False, derived=None, order="valid_from",
509
+ limit=None) -> list[Fact]:
510
+ clauses, params, branch_filter = self._fact_filters(
511
+ user_id, agent_id, run_id, branch, active, include_quarantined,
512
+ subject, relation, value, derived)
513
+ where = (" WHERE " + " AND ".join(clauses)) if clauses else ""
514
+ q = f"SELECT {FACT_COLUMNS} FROM facts{where} ORDER BY {order}"
515
+ if limit:
516
+ q += f" LIMIT {int(limit)}"
517
+ rows = self.conn.execute(q, params).fetchall()
518
+ facts = [Fact.from_row(dict(r)) for r in rows]
519
+ if branch_filter is not None:
520
+ ids = self.active_ids(branch_filter)
521
+ facts = [f for f in facts if f.id in ids]
522
+ return facts
523
+
524
+ def active_facts(self, **kw) -> list[Fact]:
525
+ return self.query_facts(**kw)
526
+
527
+ def history_of(self, subject: str, relation: str, user_id: str | None = None,
528
+ include_inactive=True) -> list[Fact]:
529
+ clauses, params, _ = self._fact_filters(
530
+ user_id=user_id, active=not include_inactive)
531
+ extra = "subject=? AND relation=?" + ((" AND " + " AND ".join(clauses)) if clauses else "")
532
+ q = f"SELECT {FACT_COLUMNS} FROM facts WHERE {extra} ORDER BY valid_from, tx_from"
533
+ rows = self.conn.execute(q, (subject, relation, *params)).fetchall()
534
+ return [Fact.from_row(dict(r)) for r in rows]
535
+
536
+ def facts_about(self, entity: str, user_id: str | None = None,
537
+ active: bool = True) -> list[Fact]:
538
+ """Facts where entity is subject OR value (1-hop associative recall)."""
539
+ clauses, params, _ = self._fact_filters(user_id=user_id, active=active)
540
+ extra = "(subject=? OR value=?)" + ((" AND " + " AND ".join(clauses)) if clauses else "")
541
+ q = f"SELECT {FACT_COLUMNS} FROM facts WHERE {extra}"
542
+ rows = self.conn.execute(q, (entity, entity, *params)).fetchall()
543
+ return [Fact.from_row(dict(r)) for r in rows]
544
+
545
+ def temporal_window(self, start: str | None, end: str | None,
546
+ user_id: str | None = None, field: str = "valid",
547
+ active: bool = True) -> list[Fact]:
548
+ """Zep-compatible temporal queries. field: 'valid' (reality) or 'tx' (recorded).
549
+
550
+ Valid-time uses interval-overlap semantics: a fact matches if its
551
+ [valid_from, valid_to] window intersects [start, end] — so asking
552
+ "where did Alice work in 2025?" retrieves an employment that began
553
+ in 2024 and ended in 2026. Transaction-time uses point semantics.
554
+ """
555
+ clauses, fparams, _ = self._fact_filters(user_id=user_id, active=active)
556
+ cond: list[str] = []
557
+ cparams: list = []
558
+ if field == "valid":
559
+ if start:
560
+ cond.append("(valid_to IS NULL OR valid_to>=?)")
561
+ cparams.append(start[:10])
562
+ if end:
563
+ cond.append("valid_from<=?")
564
+ cparams.append(end[:10])
565
+ order = "valid_from"
566
+ else:
567
+ if start:
568
+ cond.append("tx_from>=?"); cparams.append(start[:10])
569
+ if end:
570
+ cond.append("tx_from<=?"); cparams.append(end[:10])
571
+ order = "tx_from"
572
+ params = cparams + fparams
573
+ where = " AND ".join(cond + clauses) if (cond + clauses) else ""
574
+ q = f"SELECT {FACT_COLUMNS} FROM facts{' WHERE ' + where if where else ''} ORDER BY {order}"
575
+ rows = self.conn.execute(q, params).fetchall()
576
+ return [Fact.from_row(dict(r)) for r in rows]
577
+
578
+ def count_facts(self, user_id: str | None = None, active_only: bool = True) -> int:
579
+ clauses, params, _ = self._fact_filters(user_id=user_id, active=active_only)
580
+ where = (" WHERE " + " AND ".join(clauses)) if clauses else ""
581
+ row = self.conn.execute(f"SELECT COUNT(*) c FROM facts{where}", params).fetchone()
582
+ return int(row["c"])
583
+
584
+ # ---------------------------------------------------------- active set
585
+ def active_ids(self, branch: str) -> frozenset[str]:
586
+ cached = self._active_cache.get(branch)
587
+ if cached is not None:
588
+ return cached
589
+ head = self.head(branch)
590
+ if head is None:
591
+ return frozenset()
592
+ anc = self.ancestry(head)
593
+ rows = self.conn.execute(
594
+ "SELECT id, birth_commit, retired_commit FROM facts "
595
+ "WHERE birth_commit IS NOT NULL").fetchall()
596
+ ids = frozenset(
597
+ r["id"] for r in rows
598
+ if r["birth_commit"] in anc and (
599
+ not r["retired_commit"] or r["retired_commit"] not in anc))
600
+ if len(self._active_cache) < 16:
601
+ self._active_cache[branch] = ids
602
+ return ids
603
+
604
+ def _invalidate(self, branch: str | None = None) -> None:
605
+ if branch:
606
+ self._active_cache.pop(branch, None)
607
+ else:
608
+ self._active_cache.clear()
609
+ self._ancestry_cache.clear()
610
+
611
+ # ---------------------------------------------------------------- edges
612
+ def add_edge(self, src: str, dst: str, kind: str, meta: dict | None = None) -> None:
613
+ self.conn.execute(
614
+ "INSERT OR REPLACE INTO edges(src, dst, kind, meta, created) VALUES(?,?,?,?,?)",
615
+ (src, dst, kind, json.dumps(meta or {}), iso(datetime.now(timezone.utc))))
616
+ self._maybe_commit()
617
+
618
+ def edges_of(self, fact_id: str, kind: str | None = None, direction: str = "out") -> list[dict]:
619
+ outs, ins = [], []
620
+ if direction in ("out", "both"):
621
+ q = "SELECT * FROM edges WHERE src=?" + (" AND kind=?" if kind else "")
622
+ rows = self.conn.execute(q, (fact_id, kind) if kind else (fact_id,)).fetchall()
623
+ outs = [dict(r, dir="out") for r in rows]
624
+ if direction in ("in", "both"):
625
+ q = "SELECT * FROM edges WHERE dst=?" + (" AND kind=?" if kind else "")
626
+ rows = self.conn.execute(q, (fact_id, kind) if kind else (fact_id,)).fetchall()
627
+ ins = [dict(r, dir="in") for r in rows]
628
+ return outs + ins
629
+
630
+ def edges_of_many(self, fact_ids: list[str], kind: str | None = None) -> list[dict]:
631
+ """Edges among a set of fact ids (both directions), batched."""
632
+ if not fact_ids:
633
+ return []
634
+ out: list[dict] = []
635
+ B = 200
636
+ for i in range(0, len(fact_ids), B):
637
+ chunk = fact_ids[i:i + B]
638
+ qm = ",".join("?" * len(chunk))
639
+ q = (f"SELECT * FROM edges WHERE src IN ({qm}) AND dst IN ({qm})"
640
+ + (f" AND kind=?" if kind else ""))
641
+ rows = self.conn.execute(q, chunk + chunk + ([kind] if kind else []))
642
+ out.extend(dict(r) for r in rows)
643
+ return out
644
+
645
+ # ------------------------------------------------------------ integrity
646
+ def active_fact_hashes(self, user_id: str | None = None) -> list[tuple[str, str]]:
647
+ clauses, params, _ = self._fact_filters(user_id=user_id, active=True)
648
+ where = (" WHERE " + " AND ".join(clauses)) if clauses else ""
649
+ rows = self.conn.execute(
650
+ f"SELECT id, source_hash FROM facts{where} ORDER BY id", params).fetchall()
651
+ return [(r["id"], r["source_hash"] or "") for r in rows]
652
+
653
+ def stats(self) -> dict:
654
+ def one(q, *p):
655
+ return int(self.conn.execute(q, p).fetchone()[0])
656
+ return {
657
+ "facts": one("SELECT COUNT(*) FROM facts"),
658
+ "active_facts": one("SELECT COUNT(*) FROM facts WHERE is_active=1 AND quarantined=0"),
659
+ "quarantined": one("SELECT COUNT(*) FROM facts WHERE quarantined=1"),
660
+ "derived": one("SELECT COUNT(*) FROM facts WHERE is_derived=1"),
661
+ "chunks": one("SELECT COUNT(*) FROM chunks"),
662
+ "edges": one("SELECT COUNT(*) FROM edges"),
663
+ "commits": one("SELECT COUNT(*) FROM commits"),
664
+ "branches": one("SELECT COUNT(*) FROM branches"),
665
+ "long_term": one("SELECT COUNT(*) FROM facts WHERE memory_type='long_term'"),
666
+ }
667
+
668
+ def close(self) -> None:
669
+ try:
670
+ self.conn.commit()
671
+ self.checkpoint("TRUNCATE")
672
+ self.conn.close()
673
+ except Exception:
674
+ pass
675
+
676
+ def __enter__(self) -> "TraceStore":
677
+ return self
678
+
679
+ def __exit__(self, *exc) -> None:
680
+ self.close()