cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
"""Data governance: GDPR erasure, retention, backup/restore, PITR.
|
|
2
|
+
|
|
3
|
+
These are the four operations an enterprise buyer's security review
|
|
4
|
+
actually blocks on:
|
|
5
|
+
|
|
6
|
+
erase_user() — Art. 17 right-to-erasure: hard-delete every trace
|
|
7
|
+
of a subject (facts, chunks, vectors, lexicon,
|
|
8
|
+
aliases, vault entries), crypto-shred the PII vault,
|
|
9
|
+
and leave the audit chain intact (legal exemption)
|
|
10
|
+
with an erasure ATTESTATION record.
|
|
11
|
+
apply_retention() — Art. 5(1)(e) storage-limitation: expire facts and
|
|
12
|
+
chunks older than the policy window, keeping the
|
|
13
|
+
bi-temporal tombstones so history remains honest.
|
|
14
|
+
snapshot()/restore() — atomic, integrity-checked backup envelope
|
|
15
|
+
(SQLite backup API + manifest + Merkle-style digest
|
|
16
|
+
of the artifact).
|
|
17
|
+
state_at() — point-in-time recovery read: the bi-temporal Trace
|
|
18
|
+
replays "what did the system believe at T?" using
|
|
19
|
+
transaction times (tx_from/tx_to) — no snapshot
|
|
20
|
+
needed, the database IS the WAL.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import hashlib
|
|
26
|
+
import json
|
|
27
|
+
import os
|
|
28
|
+
import time
|
|
29
|
+
from datetime import datetime, timezone
|
|
30
|
+
|
|
31
|
+
from cortexm.security.hashes import HashProvider
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class Governance:
|
|
35
|
+
def __init__(self, memory) -> None:
|
|
36
|
+
self.memory = memory
|
|
37
|
+
self.store = memory.store
|
|
38
|
+
self.palace = memory.palace
|
|
39
|
+
self.audit = getattr(memory, "audit_log", None)
|
|
40
|
+
|
|
41
|
+
def _log(self, action: str, resource: str | None = None,
|
|
42
|
+
outcome: str = "success", meta: dict | None = None) -> None:
|
|
43
|
+
if self.audit is not None:
|
|
44
|
+
self.audit.log(action, resource=resource, outcome=outcome,
|
|
45
|
+
meta=meta)
|
|
46
|
+
|
|
47
|
+
# ------------------------------------------------------------ erasure
|
|
48
|
+
def erase_user(self, user_id: str, *, hard: bool = True,
|
|
49
|
+
crypto_shred: bool = True) -> dict:
|
|
50
|
+
"""GDPR Art. 17 — remove the subject from every layer."""
|
|
51
|
+
t0 = time.time()
|
|
52
|
+
conn = self.store.conn
|
|
53
|
+
counts = {}
|
|
54
|
+
counts["facts"] = conn.execute(
|
|
55
|
+
"SELECT COUNT(*) AS c FROM facts WHERE user_id=?",
|
|
56
|
+
(user_id,)).fetchone()["c"]
|
|
57
|
+
counts["chunks"] = conn.execute(
|
|
58
|
+
"SELECT COUNT(*) AS c FROM chunks WHERE user_id=?",
|
|
59
|
+
(user_id,)).fetchone()["c"]
|
|
60
|
+
# orphaned edges touching removed facts
|
|
61
|
+
fact_ids = [r["id"] for r in conn.execute(
|
|
62
|
+
"SELECT id FROM facts WHERE user_id=?", (user_id,))]
|
|
63
|
+
if fact_ids:
|
|
64
|
+
qmarks = ",".join("?" * len(fact_ids))
|
|
65
|
+
counts["edges"] = conn.execute(
|
|
66
|
+
f"SELECT COUNT(*) AS c FROM edges WHERE src IN ({qmarks}) "
|
|
67
|
+
f"OR dst IN ({qmarks})", fact_ids + fact_ids).fetchone()["c"]
|
|
68
|
+
conn.execute(f"DELETE FROM edges WHERE src IN ({qmarks}) "
|
|
69
|
+
f"OR dst IN ({qmarks})", fact_ids + fact_ids)
|
|
70
|
+
conn.execute("DELETE FROM facts WHERE user_id=?", (user_id,))
|
|
71
|
+
conn.execute("DELETE FROM chunks WHERE user_id=?", (user_id,))
|
|
72
|
+
# kv residue: lexicon, names, alias caches
|
|
73
|
+
kv_removed = 0
|
|
74
|
+
for k, _v in list(self.store.iter_kv(f"lexicon:{user_id}")):
|
|
75
|
+
self.store.kv_delete(k); kv_removed += 1
|
|
76
|
+
for k, _v in list(self.store.iter_kv(f"name:{user_id}")):
|
|
77
|
+
self.store.kv_delete(k); kv_removed += 1
|
|
78
|
+
counts["kv"] = kv_removed
|
|
79
|
+
# PII vault: crypto-shred if enabled and configured
|
|
80
|
+
vault = getattr(self.memory, "pii_vault", None)
|
|
81
|
+
if vault is not None and crypto_shred:
|
|
82
|
+
counts["vault_shredded"] = vault.crypto_shred()
|
|
83
|
+
# palace vectors for those facts
|
|
84
|
+
vec_removed = 0
|
|
85
|
+
if fact_ids:
|
|
86
|
+
vec_removed = self.palace.remove_ids(fact_ids)
|
|
87
|
+
counts["vectors"] = vec_removed
|
|
88
|
+
conn.commit()
|
|
89
|
+
# erasure attestation in the tamper-evident chain (Art. 30 records)
|
|
90
|
+
self._log("governance.erase", resource=user_id, outcome="erased",
|
|
91
|
+
meta={"counts": counts,
|
|
92
|
+
"duration_ms": round((time.time() - t0) * 1e3, 1),
|
|
93
|
+
"hard": hard, "crypto_shred": crypto_shred})
|
|
94
|
+
# residual scan: no row may reference the subject anywhere
|
|
95
|
+
residual = {
|
|
96
|
+
"facts": conn.execute(
|
|
97
|
+
"SELECT COUNT(*) AS c FROM facts WHERE user_id=?",
|
|
98
|
+
(user_id,)).fetchone()["c"],
|
|
99
|
+
"chunks": conn.execute(
|
|
100
|
+
"SELECT COUNT(*) AS c FROM chunks WHERE user_id=?",
|
|
101
|
+
(user_id,)).fetchone()["c"],
|
|
102
|
+
}
|
|
103
|
+
return {"user_id": user_id, "erased": residual["facts"] == 0
|
|
104
|
+
and residual["chunks"] == 0,
|
|
105
|
+
"counts": counts, "residual": residual,
|
|
106
|
+
"duration_ms": round((time.time() - t0) * 1e3, 1)}
|
|
107
|
+
|
|
108
|
+
# ------------------------------------------------------------ retention
|
|
109
|
+
def apply_retention(self, days: int, *, user_id: str | None = None,
|
|
110
|
+
dry_run: bool = False) -> dict:
|
|
111
|
+
"""Expire facts whose last transaction time is older than ``days``."""
|
|
112
|
+
if days <= 0:
|
|
113
|
+
raise ValueError("days must be positive")
|
|
114
|
+
cutoff = datetime.now(timezone.utc).timestamp() - days * 86400
|
|
115
|
+
cutoff_iso = datetime.fromtimestamp(cutoff, tz=timezone.utc).isoformat()
|
|
116
|
+
conn = self.store.conn
|
|
117
|
+
q = ("SELECT COUNT(*) AS c FROM facts WHERE tx_from < ? "
|
|
118
|
+
"AND tx_to IS NULL")
|
|
119
|
+
args = [cutoff_iso]
|
|
120
|
+
if user_id:
|
|
121
|
+
q += " AND user_id=?"
|
|
122
|
+
args.append(user_id)
|
|
123
|
+
stale = conn.execute(q, args).fetchone()["c"]
|
|
124
|
+
if dry_run:
|
|
125
|
+
return {"stale_facts": stale, "applied": False}
|
|
126
|
+
q2 = "UPDATE facts SET tx_to=?, is_active=0 WHERE tx_from < ? AND tx_to IS NULL"
|
|
127
|
+
args2 = [datetime.now(timezone.utc).isoformat(), cutoff_iso]
|
|
128
|
+
if user_id:
|
|
129
|
+
q2 += " AND user_id=?"
|
|
130
|
+
args2.append(user_id)
|
|
131
|
+
cur = conn.execute(q2, args2)
|
|
132
|
+
conn.commit()
|
|
133
|
+
self._log("governance.retention",
|
|
134
|
+
meta={"days": days, "expired": cur.rowcount,
|
|
135
|
+
"user_id": user_id or "all"})
|
|
136
|
+
return {"stale_facts": cur.rowcount, "applied": True,
|
|
137
|
+
"cutoff": cutoff_iso}
|
|
138
|
+
|
|
139
|
+
# ------------------------------------------------------------ snapshot
|
|
140
|
+
def snapshot(self, path: str) -> dict:
|
|
141
|
+
"""Atomic backup: online-backup the SQLite file + write a manifest.
|
|
142
|
+
The backup API copies page-by-page under a read transaction —
|
|
143
|
+
safe while writers are active."""
|
|
144
|
+
os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
|
|
145
|
+
if path.endswith("/"):
|
|
146
|
+
raise ValueError("path must be a file, not a directory")
|
|
147
|
+
target = sqlite_backup(self.store.db_path, path)
|
|
148
|
+
digest = _file_digest(target)
|
|
149
|
+
manifest = {
|
|
150
|
+
"format": "context-m-snapshot/1",
|
|
151
|
+
"created_at": datetime.now(timezone.utc).isoformat(),
|
|
152
|
+
"source_db": self.store.db_path,
|
|
153
|
+
"file": os.path.basename(target),
|
|
154
|
+
"size_bytes": os.path.getsize(target),
|
|
155
|
+
"sha256": digest,
|
|
156
|
+
"facts": self.store.conn.execute(
|
|
157
|
+
"SELECT COUNT(*) AS c FROM facts").fetchone()["c"],
|
|
158
|
+
"audit_head": (self.audit.verify()["head_hash"]
|
|
159
|
+
if self.audit else None),
|
|
160
|
+
}
|
|
161
|
+
mpath = target + ".manifest.json"
|
|
162
|
+
with open(mpath, "w", encoding="utf-8") as fh:
|
|
163
|
+
json.dump(manifest, fh, indent=2)
|
|
164
|
+
self._log("governance.snapshot",
|
|
165
|
+
resource=path, meta={"sha256": digest,
|
|
166
|
+
"size": manifest["size_bytes"]})
|
|
167
|
+
return {"path": target, "manifest": manifest, "manifest_path": mpath}
|
|
168
|
+
|
|
169
|
+
def restore(self, snapshot_path: str, *, verify: bool = True) -> dict:
|
|
170
|
+
"""Restore from a snapshot manifest pair. Refuses mismatched digests."""
|
|
171
|
+
mpath = snapshot_path + ".manifest.json"
|
|
172
|
+
if verify and not os.path.exists(mpath):
|
|
173
|
+
raise FileNotFoundError("manifest missing — cannot verify integrity")
|
|
174
|
+
if os.path.exists(mpath):
|
|
175
|
+
with open(mpath, encoding="utf-8") as fh:
|
|
176
|
+
manifest = json.load(fh)
|
|
177
|
+
if verify:
|
|
178
|
+
digest = _file_digest(snapshot_path)
|
|
179
|
+
if digest != manifest["sha256"]:
|
|
180
|
+
raise ValueError(
|
|
181
|
+
f"integrity check failed: {digest} != {manifest['sha256']}")
|
|
182
|
+
# close current handles, replace file, reopen
|
|
183
|
+
db_path = self.store.db_path
|
|
184
|
+
self.memory.close()
|
|
185
|
+
import shutil
|
|
186
|
+
shutil.copyfile(snapshot_path, db_path)
|
|
187
|
+
self.memory._reopen()
|
|
188
|
+
# _reopen() built a NEW Governance object; this method is running
|
|
189
|
+
# on the old one — rebind self so the audit attestation below
|
|
190
|
+
# writes through the fresh store, not the closed one.
|
|
191
|
+
self.store = self.memory.store
|
|
192
|
+
self.palace = self.memory.palace
|
|
193
|
+
self.audit = self.memory.audit_log
|
|
194
|
+
self._log("governance.restore", resource=snapshot_path)
|
|
195
|
+
return {"restored": db_path, "verified": verify}
|
|
196
|
+
|
|
197
|
+
# ------------------------------------------------------------ PITR
|
|
198
|
+
def state_at(self, when, *, user_id: str | None = None,
|
|
199
|
+
limit: int = 500) -> list[dict]:
|
|
200
|
+
"""Point-in-time read: facts the system believed true at ``when``
|
|
201
|
+
(transaction-time replay — the database is its own WAL)."""
|
|
202
|
+
from cortexm.api.memory import parse_ts
|
|
203
|
+
ts = parse_ts(when) if not isinstance(when, datetime) else when
|
|
204
|
+
if ts.tzinfo is None:
|
|
205
|
+
ts = ts.replace(tzinfo=timezone.utc)
|
|
206
|
+
# normalize to the Z-suffix format facts are stored with, so the
|
|
207
|
+
# string comparison in SQL matches the stored tx_from exactly
|
|
208
|
+
iso = ts.strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
209
|
+
q = ("SELECT * FROM facts WHERE tx_from <= ? AND "
|
|
210
|
+
"(tx_to IS NULL OR tx_to > ?) AND quarantined=0")
|
|
211
|
+
args: list = [iso, iso]
|
|
212
|
+
if user_id:
|
|
213
|
+
q += " AND user_id=?"
|
|
214
|
+
args.append(user_id)
|
|
215
|
+
q += " ORDER BY valid_from DESC LIMIT ?"
|
|
216
|
+
args.append(limit)
|
|
217
|
+
rows = [dict(r) for r in self.store.conn.execute(q, args)]
|
|
218
|
+
self._log("governance.pitr", meta={"when": iso, "rows": len(rows)})
|
|
219
|
+
return rows
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
# ------------------------------------------------------------------ helpers
|
|
223
|
+
def sqlite_backup(src_db: str, dst_path: str) -> str:
|
|
224
|
+
import sqlite3
|
|
225
|
+
src = sqlite3.connect(src_db)
|
|
226
|
+
dst = sqlite3.connect(dst_path)
|
|
227
|
+
with dst:
|
|
228
|
+
src.backup(dst)
|
|
229
|
+
dst.close()
|
|
230
|
+
src.close()
|
|
231
|
+
return dst_path
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _file_digest(path: str) -> str:
|
|
235
|
+
h = hashlib.sha256()
|
|
236
|
+
with open(path, "rb") as fh:
|
|
237
|
+
for chunk in iter(lambda: fh.read(1 << 20), b""):
|
|
238
|
+
h.update(chunk)
|
|
239
|
+
return h.hexdigest()
|
cortexm/errors.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Exception hierarchy for Context-M."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class ContextMError(Exception):
|
|
7
|
+
"""Base class."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class StoreError(ContextMError):
|
|
11
|
+
"""Trace / SQLite layer failure."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ExtractionError(ContextMError):
|
|
15
|
+
"""Deterministic extraction pipeline failure."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class SecurityError(ContextMError):
|
|
19
|
+
"""Provenance / hash verification failure."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class VerificationError(SecurityError):
|
|
23
|
+
"""Hash or Merkle proof did not verify."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class BranchError(ContextMError):
|
|
27
|
+
"""Memory-Git operation failure (unknown branch, dirty state...)."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class MigrationError(ContextMError):
|
|
31
|
+
"""Import from a foreign memory system failed."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class CodecError(ContextMError):
|
|
35
|
+
"""Vector codec failure (untrained PQ, corrupt record...)."""
|
|
File without changes
|
cortexm/features/git.py
ADDED
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""Memory Git — version control for agent memory.
|
|
2
|
+
|
|
3
|
+
Every add() is a hash-chained commit on a Merkle-style DAG. Branches
|
|
4
|
+
fork memory state (A/B-test an agent personality), merges replay
|
|
5
|
+
theirs-onto-ours with deterministic conflict resolution via the
|
|
6
|
+
lifecycle engine, diffs show exactly which facts changed, and blame
|
|
7
|
+
traces which commit introduced a fact. Enterprises get regulatory
|
|
8
|
+
rollback and forensic audit as first-class operations.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
|
|
15
|
+
from cortexm.errors import BranchError
|
|
16
|
+
from cortexm.trace.fact import Fact, SINGLE_VALUED
|
|
17
|
+
from cortexm.trace.store import TraceStore
|
|
18
|
+
from cortexm.util import new_id, similarity
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class MemoryGit:
|
|
22
|
+
def __init__(self, store: TraceStore, palace=None) -> None:
|
|
23
|
+
self.store = store
|
|
24
|
+
self.palace = palace
|
|
25
|
+
|
|
26
|
+
# ------------------------------------------------------------- basics
|
|
27
|
+
def branch(self, name: str, from_commit: str | None = None,
|
|
28
|
+
switch: bool = True) -> str:
|
|
29
|
+
base = self.store.create_branch(name, from_commit, switch=switch)
|
|
30
|
+
return base
|
|
31
|
+
|
|
32
|
+
def checkout(self, name: str) -> None:
|
|
33
|
+
self.store.checkout(name)
|
|
34
|
+
|
|
35
|
+
def branches(self) -> list[dict]:
|
|
36
|
+
return self.store.branches()
|
|
37
|
+
|
|
38
|
+
def log(self, branch: str | None = None, limit: int = 50) -> list[dict]:
|
|
39
|
+
return self.store.log(branch, limit)
|
|
40
|
+
|
|
41
|
+
# ------------------------------------------------------------- sets
|
|
42
|
+
def active_at(self, commit_id: str) -> dict[str, Fact]:
|
|
43
|
+
anc = self.store.ancestry(commit_id)
|
|
44
|
+
rows = self.store.conn.execute(
|
|
45
|
+
"SELECT id, birth_commit, retired_commit FROM facts "
|
|
46
|
+
"WHERE birth_commit IS NOT NULL").fetchall()
|
|
47
|
+
ids = [r["id"] for r in rows
|
|
48
|
+
if r["birth_commit"] in anc
|
|
49
|
+
and (not r["retired_commit"] or r["retired_commit"] not in anc)]
|
|
50
|
+
return {f.id: f for f in self.store.get_facts(ids)}
|
|
51
|
+
|
|
52
|
+
def _triple_key(self, f: Fact) -> tuple:
|
|
53
|
+
return (f.subject, f.relation, f.value.lower())
|
|
54
|
+
|
|
55
|
+
# ------------------------------------------------------------- diff
|
|
56
|
+
def diff(self, a: str, b: str) -> dict:
|
|
57
|
+
fa, fb = self.active_at(a), self.active_at(b)
|
|
58
|
+
ka = {self._triple_key(f): f for f in fa.values()}
|
|
59
|
+
kb = {self._triple_key(f): f for f in fb.values()}
|
|
60
|
+
added = [kb[k] for k in kb.keys() - ka.keys()]
|
|
61
|
+
removed = [ka[k] for k in ka.keys() - kb.keys()]
|
|
62
|
+
return {
|
|
63
|
+
"from": a, "to": b,
|
|
64
|
+
"added": [{"id": f.id, "fact": f.display(),
|
|
65
|
+
"valid_from": f.valid_from} for f in added],
|
|
66
|
+
"removed": [{"id": f.id, "fact": f.display(),
|
|
67
|
+
"valid_from": f.valid_from} for f in removed],
|
|
68
|
+
"n_added": len(added), "n_removed": len(removed),
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
# ------------------------------------------------------------- blame
|
|
72
|
+
def blame(self, subject: str, relation: str | None = None,
|
|
73
|
+
user_id: str | None = None) -> list[dict]:
|
|
74
|
+
clauses = ["subject=?"]
|
|
75
|
+
params: list = [subject]
|
|
76
|
+
if relation:
|
|
77
|
+
clauses.append("relation=?")
|
|
78
|
+
params.append(relation)
|
|
79
|
+
if user_id:
|
|
80
|
+
clauses.append("user_id=?")
|
|
81
|
+
params.append(user_id)
|
|
82
|
+
rows = self.store.conn.execute(
|
|
83
|
+
f"SELECT id, relation, value, valid_from, valid_to, is_active, "
|
|
84
|
+
f"birth_commit, tx_from FROM facts WHERE {' AND '.join(clauses)} "
|
|
85
|
+
f"ORDER BY tx_from", params).fetchall()
|
|
86
|
+
out = []
|
|
87
|
+
for r in rows:
|
|
88
|
+
commit = self.store.commit(r["birth_commit"]) if r["birth_commit"] else None
|
|
89
|
+
out.append({
|
|
90
|
+
"fact_id": r["id"],
|
|
91
|
+
"fact": f"({subject}, {r['relation']}, {r['value']})",
|
|
92
|
+
"valid": f"{r['valid_from']}→{r['valid_to'] or '∞'}",
|
|
93
|
+
"active": bool(r["is_active"]),
|
|
94
|
+
"commit": r["birth_commit"],
|
|
95
|
+
"commit_message": commit["message"] if commit else None,
|
|
96
|
+
"recorded_at": r["tx_from"],
|
|
97
|
+
})
|
|
98
|
+
return out
|
|
99
|
+
|
|
100
|
+
# ------------------------------------------------------------- merge
|
|
101
|
+
def merge(self, name: str, strategy: str = "latest-wins",
|
|
102
|
+
message: str = "") -> dict:
|
|
103
|
+
"""3-way merge of branch ``name`` into the current branch."""
|
|
104
|
+
if strategy not in ("latest-wins", "union"):
|
|
105
|
+
raise BranchError(f"unknown strategy {strategy!r}")
|
|
106
|
+
cur = self.store.current_branch()
|
|
107
|
+
ours_head = self.store.head(cur)
|
|
108
|
+
theirs_head = self.store.head(name)
|
|
109
|
+
if not theirs_head:
|
|
110
|
+
raise BranchError(f"branch {name!r} has no commits")
|
|
111
|
+
if theirs_head == ours_head:
|
|
112
|
+
return {"status": "already-merged", "commit": ours_head,
|
|
113
|
+
"applied": 0, "conflicts": 0}
|
|
114
|
+
|
|
115
|
+
ours = self.active_at(ours_head)
|
|
116
|
+
theirs = self.active_at(theirs_head)
|
|
117
|
+
base_head = self._common_ancestor(ours_head, theirs_head)
|
|
118
|
+
base = self.active_at(base_head) if base_head else {}
|
|
119
|
+
|
|
120
|
+
kb = {self._triple_key(f): f for f in base.values()}
|
|
121
|
+
ko = {self._triple_key(f): f for f in ours.values()}
|
|
122
|
+
kt = {self._triple_key(f): f for f in theirs.values()}
|
|
123
|
+
|
|
124
|
+
added_by_theirs = [kt[k] for k in kt.keys() - kb.keys()]
|
|
125
|
+
retired_by_theirs = [kb[k] for k in kb.keys() - kt.keys()]
|
|
126
|
+
|
|
127
|
+
merge_commit = self.store.create_commit(
|
|
128
|
+
message or f"merge {name} into {cur} ({strategy})",
|
|
129
|
+
branch=cur, parents=[ours_head, theirs_head])
|
|
130
|
+
applied, conflicts = 0, 0
|
|
131
|
+
|
|
132
|
+
ours_by_sr: dict[tuple, list[Fact]] = {}
|
|
133
|
+
for f in ours.values():
|
|
134
|
+
ours_by_sr.setdefault((f.subject, f.relation), []).append(f)
|
|
135
|
+
|
|
136
|
+
for f in added_by_theirs:
|
|
137
|
+
key = self._triple_key(f)
|
|
138
|
+
if key in ko:
|
|
139
|
+
continue # already present on our side
|
|
140
|
+
conflict = any(g.value.lower() != f.value.lower()
|
|
141
|
+
for g in ours_by_sr.get((f.subject, f.relation), [])
|
|
142
|
+
if g.is_active)
|
|
143
|
+
if conflict and f.relation in SINGLE_VALUED and strategy == "latest-wins":
|
|
144
|
+
for g in ours_by_sr.get((f.subject, f.relation), []):
|
|
145
|
+
if not g.is_active:
|
|
146
|
+
continue
|
|
147
|
+
if (g.tx_from or "") <= (f.tx_from or ""):
|
|
148
|
+
self.store.update_fact(
|
|
149
|
+
g.id, is_active=0, retired_commit=merge_commit,
|
|
150
|
+
tx_to=f.tx_from,
|
|
151
|
+
provenance={**g.provenance,
|
|
152
|
+
"merged_away": merge_commit})
|
|
153
|
+
conflicts += 1
|
|
154
|
+
self._copy_fact(f, merge_commit)
|
|
155
|
+
applied += 1
|
|
156
|
+
elif conflict:
|
|
157
|
+
self._copy_fact(f, merge_commit) # union: keep both
|
|
158
|
+
conflicts += 1
|
|
159
|
+
applied += 1
|
|
160
|
+
else:
|
|
161
|
+
self._copy_fact(f, merge_commit)
|
|
162
|
+
applied += 1
|
|
163
|
+
|
|
164
|
+
for f in retired_by_theirs:
|
|
165
|
+
key = self._triple_key(f)
|
|
166
|
+
if key in ko and ko[key].is_active:
|
|
167
|
+
self.store.update_fact(
|
|
168
|
+
ko[key].id, is_active=0, retired_commit=merge_commit,
|
|
169
|
+
provenance={**ko[key].provenance,
|
|
170
|
+
"merged_away": f"retired in {name}"})
|
|
171
|
+
applied += 1
|
|
172
|
+
|
|
173
|
+
self.store.conn.commit()
|
|
174
|
+
return {"status": "merged", "commit": merge_commit,
|
|
175
|
+
"base": base_head, "applied": applied, "conflicts": conflicts,
|
|
176
|
+
"strategy": strategy}
|
|
177
|
+
|
|
178
|
+
def _copy_fact(self, f: Fact, commit_id: str) -> Fact:
|
|
179
|
+
nf = Fact(
|
|
180
|
+
id=new_id(), subject=f.subject, relation=f.relation, value=f.value,
|
|
181
|
+
valid_from=f.valid_from, valid_to=f.valid_to, tx_from=f.tx_from,
|
|
182
|
+
confidence=f.confidence, source_hash=f.source_hash,
|
|
183
|
+
source_id=f.source_id, user_id=f.user_id, agent_id=f.agent_id,
|
|
184
|
+
run_id=f.run_id, memory_type=f.memory_type,
|
|
185
|
+
access_count=f.access_count, reinforcement=f.reinforcement,
|
|
186
|
+
is_active=True, is_derived=f.is_derived,
|
|
187
|
+
provenance={**f.provenance, "merged_from": f.id})
|
|
188
|
+
self.store.insert_fact(nf, commit_id)
|
|
189
|
+
if self.palace is not None:
|
|
190
|
+
self.palace.add(nf.id, self.palace.encode_fact(nf))
|
|
191
|
+
return nf
|
|
192
|
+
|
|
193
|
+
def _common_ancestor(self, a: str, b: str) -> str | None:
|
|
194
|
+
anc_a = self.store.ancestry(a)
|
|
195
|
+
anc_b = self.store.ancestry(b)
|
|
196
|
+
common = anc_a & anc_b
|
|
197
|
+
if not common:
|
|
198
|
+
return None
|
|
199
|
+
best, best_ts = None, ""
|
|
200
|
+
for cid in common:
|
|
201
|
+
c = self.store.commit(cid)
|
|
202
|
+
if c and c["ts"] > best_ts:
|
|
203
|
+
best, best_ts = cid, c["ts"]
|
|
204
|
+
return best
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Predictive Memory Prefetching — the Memory Branch Target Buffer.
|
|
2
|
+
|
|
3
|
+
Learns co-access patterns across retrievals ("agents that asked about X
|
|
4
|
+
next asked about Y") and prefetches predicted facts into the fusion
|
|
5
|
+
boost set before the next query lands — branch prediction for agent
|
|
6
|
+
memory. Wrong predictions cost nothing (a tiny score boost); right
|
|
7
|
+
predictions cut effective retrieval latency toward the SLB hit path.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class Prefetcher:
|
|
14
|
+
def __init__(self, max_pairs: int = 200_000, decay: float = 0.98,
|
|
15
|
+
min_weight: float = 0.05, window: int = 6) -> None:
|
|
16
|
+
self.max_pairs = max_pairs
|
|
17
|
+
self.decay = decay
|
|
18
|
+
self.min_weight = min_weight
|
|
19
|
+
self.window = window
|
|
20
|
+
self._by_fid: dict[str, dict[str, float]] = {}
|
|
21
|
+
self._recent: list[str] = []
|
|
22
|
+
self._last_predict: dict[str, float] = {}
|
|
23
|
+
self._total_pairs = 0
|
|
24
|
+
self.hits = 0
|
|
25
|
+
self.predictions = 0
|
|
26
|
+
self.predicted_total = 0
|
|
27
|
+
|
|
28
|
+
# ------------------------------------------------------------------
|
|
29
|
+
def _bump(self, a: str, b: str) -> None:
|
|
30
|
+
row = self._by_fid.setdefault(a, {})
|
|
31
|
+
prev = row.get(b, 0.0)
|
|
32
|
+
row[b] = prev + 1.0
|
|
33
|
+
if prev == 0.0:
|
|
34
|
+
self._total_pairs += 1
|
|
35
|
+
|
|
36
|
+
def observe(self, fact_ids: list[str]) -> None:
|
|
37
|
+
"""Called after each retrieval with the delivered fact set."""
|
|
38
|
+
combined = list(dict.fromkeys(
|
|
39
|
+
(self._recent[-self.window:] or []) + list(fact_ids)))
|
|
40
|
+
for i, a in enumerate(combined):
|
|
41
|
+
for b in combined[i + 1:i + 5]:
|
|
42
|
+
if a != b:
|
|
43
|
+
self._bump(a, b)
|
|
44
|
+
self._bump(b, a)
|
|
45
|
+
self._recent = list(fact_ids)
|
|
46
|
+
if self._total_pairs > self.max_pairs:
|
|
47
|
+
self._prune()
|
|
48
|
+
|
|
49
|
+
def _prune(self) -> None:
|
|
50
|
+
for fid in list(self._by_fid):
|
|
51
|
+
row = self._by_fid[fid]
|
|
52
|
+
for other in list(row):
|
|
53
|
+
row[other] *= self.decay
|
|
54
|
+
if row[other] < self.min_weight:
|
|
55
|
+
del row[other]
|
|
56
|
+
self._total_pairs -= 1
|
|
57
|
+
if not row:
|
|
58
|
+
del self._by_fid[fid]
|
|
59
|
+
|
|
60
|
+
# ------------------------------------------------------------------
|
|
61
|
+
def predict(self) -> dict[str, float]:
|
|
62
|
+
"""Predict next-access facts from recent history (MBTB lookup)."""
|
|
63
|
+
out: dict[str, float] = {}
|
|
64
|
+
for fid in self._recent[-self.window:]:
|
|
65
|
+
row = self._by_fid.get(fid)
|
|
66
|
+
if not row:
|
|
67
|
+
continue
|
|
68
|
+
for other, w in row.items():
|
|
69
|
+
if w >= 1.0:
|
|
70
|
+
out[other] = max(out.get(other, 0.0), min(1.0, w / 8.0))
|
|
71
|
+
self._last_predict = out
|
|
72
|
+
self.predictions += 1
|
|
73
|
+
self.predicted_total += len(out)
|
|
74
|
+
return out
|
|
75
|
+
|
|
76
|
+
def note_hits(self, delivered_ids: list[str]) -> int:
|
|
77
|
+
hits = sum(1 for fid in delivered_ids if fid in self._last_predict)
|
|
78
|
+
self.hits += hits
|
|
79
|
+
return hits
|
|
80
|
+
|
|
81
|
+
def stats(self) -> dict:
|
|
82
|
+
return {
|
|
83
|
+
"pairs": self._total_pairs,
|
|
84
|
+
"predictions": self.predictions,
|
|
85
|
+
"prefetch_hits": self.hits,
|
|
86
|
+
"prefetch_hit_ratio": round(self.hits / self.predicted_total, 4)
|
|
87
|
+
if self.predicted_total else 0.0,
|
|
88
|
+
}
|
cortexm/features/zk.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Zero-Knowledge Memory Proofs (ZK-lite).
|
|
2
|
+
|
|
3
|
+
Proves a retrieved fact satisfies a query WITHOUT revealing its content
|
|
4
|
+
to the LLM: a Merkle membership proof over the tamper-evident leaf set
|
|
5
|
+
(blake3 source hashes) plus an HMAC attestation binding
|
|
6
|
+
{statement, root, timestamp}. The LLM receives only
|
|
7
|
+
``[ZK-Proof: match on <relation> verified. Content redacted.]``.
|
|
8
|
+
|
|
9
|
+
Honest scope: this is a commit-and-prove membership attestation — full
|
|
10
|
+
ZK-SNARKs over the similarity predicate are on the roadmap (the binary
|
|
11
|
+
codec's Hamming-distance similarity is a natural circuit candidate;
|
|
12
|
+
HRR circular convolution is a group operation, the algebraic
|
|
13
|
+
requirement standard cosine similarity cannot satisfy).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import secrets
|
|
19
|
+
import time
|
|
20
|
+
|
|
21
|
+
from cortexm.errors import VerificationError
|
|
22
|
+
from cortexm.security.hashes import (HashProvider, attest, merkle_proof,
|
|
23
|
+
merkle_verify, verify_attest)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ZKProver:
|
|
27
|
+
def __init__(self, store, reader, provider: HashProvider | None = None) -> None:
|
|
28
|
+
self.store = store
|
|
29
|
+
self.reader = reader
|
|
30
|
+
self.hasher = provider or store.hasher
|
|
31
|
+
key = self.store.kv_get("ZK_KEY")
|
|
32
|
+
if not key:
|
|
33
|
+
key = secrets.token_hex(32)
|
|
34
|
+
self.store.kv_set("ZK_KEY", key)
|
|
35
|
+
self._key = bytes.fromhex(key)
|
|
36
|
+
self._leaf_cache: tuple[str, list[str]] = ("", [])
|
|
37
|
+
|
|
38
|
+
# ------------------------------------------------------------------
|
|
39
|
+
def _leaves(self) -> list[str]:
|
|
40
|
+
head = self.store.head() or ""
|
|
41
|
+
if self._leaf_cache[0] != head or not self._leaf_cache[1]:
|
|
42
|
+
hashes = self.store.active_fact_hashes()
|
|
43
|
+
leaves = [f"{fid}:{h}" for fid, h in hashes]
|
|
44
|
+
# leaf hash -> hex digest for merkle
|
|
45
|
+
leaves = [self.hasher.hash_text(l) for l in leaves]
|
|
46
|
+
self._leaf_cache = (head, leaves)
|
|
47
|
+
return self._leaf_cache[1]
|
|
48
|
+
|
|
49
|
+
# ------------------------------------------------------------------
|
|
50
|
+
def prove(self, query: str, *, user_id: str = "default",
|
|
51
|
+
threshold: float = 0.2) -> dict:
|
|
52
|
+
"""Retrieve top match and produce a content-free proof."""
|
|
53
|
+
result = self.reader.search(query, user_id=user_id, k=1)
|
|
54
|
+
if not result.facts:
|
|
55
|
+
raise VerificationError("no matching fact to prove")
|
|
56
|
+
f = result.facts[0]
|
|
57
|
+
leaves = self._leaves()
|
|
58
|
+
idx = None
|
|
59
|
+
want = self.hasher.hash_text(f"{f.id}:{f.source_hash}")
|
|
60
|
+
for i, leaf in enumerate(leaves):
|
|
61
|
+
if leaf == want:
|
|
62
|
+
idx = i
|
|
63
|
+
break
|
|
64
|
+
if idx is None:
|
|
65
|
+
raise VerificationError("fact not in active leaf set")
|
|
66
|
+
root, path = merkle_proof(self.hasher, leaves, idx)
|
|
67
|
+
score = result.scores.get(f.id, 0.0)
|
|
68
|
+
if score < threshold:
|
|
69
|
+
raise VerificationError(
|
|
70
|
+
f"similarity {score:.3f} below threshold {threshold}")
|
|
71
|
+
statement = (f"EXISTS fact f in Trace: sim(query, f) >= {threshold} "
|
|
72
|
+
f"AND blake3(source(f)) = {f.source_hash[:16]}... "
|
|
73
|
+
f"(content redacted)")
|
|
74
|
+
ts = time.time()
|
|
75
|
+
tag = attest(self.hasher, self._key, f"{statement}|{root}|{ts}")
|
|
76
|
+
return {
|
|
77
|
+
"statement": statement,
|
|
78
|
+
"fact_commitment": f.source_hash, # content-free commitment
|
|
79
|
+
"leaf_commitment": want, # merkle leaf hash
|
|
80
|
+
"relation": f.relation, # safe to disclose
|
|
81
|
+
"sim_score": round(score, 4),
|
|
82
|
+
"merkle_root": root,
|
|
83
|
+
"merkle_path": path,
|
|
84
|
+
"timestamp": ts,
|
|
85
|
+
"attestation": tag,
|
|
86
|
+
"llm_view": f"[ZK-Proof: high-confidence match on '{f.relation}' "
|
|
87
|
+
f"verified (score {score:.2f}). Content redacted.]",
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
# ------------------------------------------------------------------
|
|
91
|
+
def verify(self, proof: dict) -> bool:
|
|
92
|
+
"""Verify membership + attestation. Returns True if sound."""
|
|
93
|
+
if not verify_attest(self.hasher, self._key,
|
|
94
|
+
f"{proof['statement']}|{proof['merkle_root']}|"
|
|
95
|
+
f"{proof['timestamp']}",
|
|
96
|
+
proof["attestation"]):
|
|
97
|
+
return False
|
|
98
|
+
leaf = proof.get("leaf_commitment") or proof["fact_commitment"]
|
|
99
|
+
return merkle_verify(self.hasher, leaf,
|
|
100
|
+
proof["merkle_path"], proof["merkle_root"])
|
|
101
|
+
|
|
102
|
+
def verify_membership(self, leaf_hex: str, proof: dict) -> bool:
|
|
103
|
+
"""Full membership verification for a known leaf commitment."""
|
|
104
|
+
return merkle_verify(self.hasher, leaf_hex, proof["merkle_path"],
|
|
105
|
+
proof["merkle_root"])
|